From 7f0c4702516387cdc4d1e7a46274af1a873904b2 Mon Sep 17 00:00:00 2001 From: riscvcxh1 Date: Wed, 26 Aug 2026 10:39:02 +0000 Subject: [PATCH] R: 4 failed research runs on round. --- research/.xh1/logs/harness.log | 92 ++++ research/.xh1/runs/20260826T101823Z/attempts | 8 + research/.xh1/runs/20260826T101823Z/completed | 0 research/.xh1/runs/20260826T101823Z/failed | 1 + ...rification_core-verification.md.current.md | 385 +++++++++++++++++ ...erification_core-verification.md.review.md | 71 ++++ ...cation_core-verification.md.review.md.json | 129 ++++++ ...erification_core-verification.md.round1.md | 293 +++++++++++++ ...cation_core-verification.md.round1.md.json | 327 ++++++++++++++ ...erification_core-verification.md.round2.md | 342 +++++++++++++++ ...cation_core-verification.md.round2.md.json | 301 +++++++++++++ ...erification_core-verification.md.round3.md | 370 ++++++++++++++++ ...cation_core-verification.md.round3.md.json | 285 +++++++++++++ ...erification_core-verification.md.round4.md | 385 +++++++++++++++++ ...cation_core-verification.md.round4.md.json | 353 +++++++++++++++ research/.xh1/runs/20260826T102850Z/attempts | 8 + research/.xh1/runs/20260826T102850Z/completed | 0 research/.xh1/runs/20260826T102850Z/failed | 1 + ...rification_core-verification.md.current.md | 402 ++++++++++++++++++ ...erification_core-verification.md.review.md | 60 +++ ...cation_core-verification.md.review.md.json | 157 +++++++ ...erification_core-verification.md.round1.md | 107 +++++ ...cation_core-verification.md.round1.md.json | 113 +++++ ...erification_core-verification.md.round2.md | 316 ++++++++++++++ ...cation_core-verification.md.round2.md.json | 383 +++++++++++++++++ ...erification_core-verification.md.round3.md | 353 +++++++++++++++ ...cation_core-verification.md.round3.md.json | 223 ++++++++++ ...erification_core-verification.md.round4.md | 402 ++++++++++++++++++ ...cation_core-verification.md.round4.md.json | 297 +++++++++++++ 29 files changed, 6164 insertions(+) create mode 100644 research/.xh1/runs/20260826T101823Z/attempts create mode 100644 research/.xh1/runs/20260826T101823Z/completed create mode 100644 research/.xh1/runs/20260826T101823Z/failed create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.current.md create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.review.md create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.review.md.json create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round1.md create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round1.md.json create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round2.md create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round2.md.json create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round3.md create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round3.md.json create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round4.md create mode 100644 research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round4.md.json create mode 100644 research/.xh1/runs/20260826T102850Z/attempts create mode 100644 research/.xh1/runs/20260826T102850Z/completed create mode 100644 research/.xh1/runs/20260826T102850Z/failed create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.current.md create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.review.md create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.review.md.json create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round1.md create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round1.md.json create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round2.md create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round2.md.json create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round3.md create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round3.md.json create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round4.md create mode 100644 research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round4.md.json diff --git a/research/.xh1/logs/harness.log b/research/.xh1/logs/harness.log index b58f0b5..3455c54 100644 --- a/research/.xh1/logs/harness.log +++ b/research/.xh1/logs/harness.log @@ -228,3 +228,95 @@ [2026-08-25T18:43:06Z] Research round 1/4 [2026-08-25T18:43:06Z] Running researcher. [2026-08-25T18:43:06Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:18:23Z] Started run: 20260826T101823Z +[2026-08-26T10:18:23Z] ================================================== +[2026-08-26T10:18:23Z] Researching: research/10-verification/core-verification.md +[2026-08-26T10:18:23Z] ================================================== +[2026-08-26T10:18:23Z] Research round 1/4 +[2026-08-26T10:18:23Z] Running researcher. +[2026-08-26T10:18:23Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:19:32Z] API response received: 28678 bytes +[2026-08-26T10:19:32Z] Running reviewer. +[2026-08-26T10:19:32Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:20:02Z] API response received: 14571 bytes +[2026-08-26T10:20:02Z] Research rejected by reviewer. +[2026-08-26T10:20:02Z] Preparing revision round 2. +[2026-08-26T10:20:07Z] Research round 2/4 +[2026-08-26T10:20:07Z] Running revision agent. +[2026-08-26T10:20:07Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:21:10Z] API response received: 41424 bytes +[2026-08-26T10:21:10Z] Running reviewer. +[2026-08-26T10:21:10Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:21:41Z] API response received: 13254 bytes +[2026-08-26T10:21:41Z] Research rejected by reviewer. +[2026-08-26T10:21:41Z] Preparing revision round 3. +[2026-08-26T10:21:46Z] Research round 3/4 +[2026-08-26T10:21:46Z] Running revision agent. +[2026-08-26T10:21:46Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:22:46Z] API response received: 53652 bytes +[2026-08-26T10:22:46Z] Running reviewer. +[2026-08-26T10:22:46Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:22:48Z] API request failed. +[2026-08-26T10:22:49Z] API request: model=minimax/minimax-m3:free attempt=2 +[2026-08-26T10:23:10Z] API response received: 9256 bytes +[2026-08-26T10:23:10Z] Research rejected by reviewer. +[2026-08-26T10:23:10Z] Preparing revision round 4. +[2026-08-26T10:23:15Z] Research round 4/4 +[2026-08-26T10:23:15Z] Running revision agent. +[2026-08-26T10:23:15Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:24:29Z] API response received: 60501 bytes +[2026-08-26T10:24:30Z] Running reviewer. +[2026-08-26T10:24:30Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:24:57Z] API response received: 11816 bytes +[2026-08-26T10:24:57Z] Research rejected by reviewer. +[2026-08-26T10:24:57Z] Maximum research rounds reached. +[2026-08-26T10:24:57Z] Leaving original document unchanged. +[2026-08-26T10:24:57Z] Marking topic failed for this run. +[2026-08-26T10:28:50Z] Started run: 20260826T102850Z +[2026-08-26T10:28:50Z] ================================================== +[2026-08-26T10:28:50Z] Researching: research/10-verification/core-verification.md +[2026-08-26T10:28:50Z] ================================================== +[2026-08-26T10:28:50Z] Research round 1/4 +[2026-08-26T10:28:50Z] Running researcher. +[2026-08-26T10:28:50Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:29:14Z] API response received: 8657 bytes +[2026-08-26T10:29:14Z] Running reviewer. +[2026-08-26T10:29:14Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:29:29Z] API response received: 5410 bytes +[2026-08-26T10:29:29Z] Research rejected by reviewer. +[2026-08-26T10:29:29Z] Preparing revision round 2. +[2026-08-26T10:29:34Z] Research round 2/4 +[2026-08-26T10:29:34Z] Running revision agent. +[2026-08-26T10:29:34Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:29:35Z] API request failed. +[2026-08-26T10:29:36Z] API request: model=minimax/minimax-m3:free attempt=2 +[2026-08-26T10:30:56Z] API response received: 27342 bytes +[2026-08-26T10:30:56Z] Running reviewer. +[2026-08-26T10:30:56Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:31:37Z] API response received: 17934 bytes +[2026-08-26T10:31:37Z] Research rejected by reviewer. +[2026-08-26T10:31:37Z] Preparing revision round 3. +[2026-08-26T10:31:42Z] Research round 3/4 +[2026-08-26T10:31:42Z] Running revision agent. +[2026-08-26T10:31:42Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:32:30Z] API response received: 35961 bytes +[2026-08-26T10:32:30Z] Running reviewer. +[2026-08-26T10:32:30Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:32:31Z] API request failed. +[2026-08-26T10:32:32Z] API request: model=minimax/minimax-m3:free attempt=2 +[2026-08-26T10:32:33Z] API request failed. +[2026-08-26T10:32:37Z] API request: model=minimax/minimax-m3:free attempt=3 +[2026-08-26T10:33:05Z] API response received: 11210 bytes +[2026-08-26T10:33:05Z] Research rejected by reviewer. +[2026-08-26T10:33:05Z] Preparing revision round 4. +[2026-08-26T10:33:10Z] Research round 4/4 +[2026-08-26T10:33:10Z] Running revision agent. +[2026-08-26T10:33:10Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:34:13Z] API response received: 48686 bytes +[2026-08-26T10:34:13Z] Running reviewer. +[2026-08-26T10:34:13Z] API request: model=minimax/minimax-m3:free attempt=1 +[2026-08-26T10:34:46Z] API response received: 13487 bytes +[2026-08-26T10:34:46Z] Research rejected by reviewer. +[2026-08-26T10:34:46Z] Maximum research rounds reached. +[2026-08-26T10:34:46Z] Leaving original document unchanged. +[2026-08-26T10:34:46Z] Marking topic failed for this run. diff --git a/research/.xh1/runs/20260826T101823Z/attempts b/research/.xh1/runs/20260826T101823Z/attempts new file mode 100644 index 0000000..fca4613 --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/attempts @@ -0,0 +1,8 @@ +2026-08-26T10:19:32Z research/10-verification/core-verification.md 1 research completed +2026-08-26T10:20:02Z research/10-verification/core-verification.md 1 review VERDICT: FAIL +2026-08-26T10:21:10Z research/10-verification/core-verification.md 2 revision completed +2026-08-26T10:21:41Z research/10-verification/core-verification.md 2 review VERDICT: FAIL +2026-08-26T10:22:46Z research/10-verification/core-verification.md 3 revision completed +2026-08-26T10:23:10Z research/10-verification/core-verification.md 3 review VERDICT: FAIL +2026-08-26T10:24:29Z research/10-verification/core-verification.md 4 revision completed +2026-08-26T10:24:57Z research/10-verification/core-verification.md 4 review VERDICT: FAIL diff --git a/research/.xh1/runs/20260826T101823Z/completed b/research/.xh1/runs/20260826T101823Z/completed new file mode 100644 index 0000000..e69de29 diff --git a/research/.xh1/runs/20260826T101823Z/failed b/research/.xh1/runs/20260826T101823Z/failed new file mode 100644 index 0000000..c6cfc59 --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/failed @@ -0,0 +1 @@ +research/10-verification/core-verification.md diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.current.md b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.current.md new file mode 100644 index 0000000..2035a2b --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.current.md @@ -0,0 +1,385 @@ +# Core Verification + +## Status + +DRAFT — Research framework established; all claims are explicitly classified by epistemic status (FACT with citation, ASSUMPTION, PROPOSAL, RECOMMENDATION, OPEN QUESTION, or INSUFFICIENT EVIDENCE). Concrete sign-off criteria, tool selection, and quantitative thresholds are deferred until the OPEN QUESTION inputs are resolved. The numeric thresholds appearing in the Sign-Off Criteria section are illustrative starting points for the verification plan and are not validated targets. Every threshold below is restated with this caveat at the point of use so that the sign-off list is internally consistent with this status. + +## Abstract + +This document investigates verification strategies for the XH-1 core, defined here as a single processing core instance within a custom 128-core RISC-V processor. Core verification is the layer of the verification methodology stack concerned with demonstrating that one core conforms to its architectural specification, microarchitectural contract, and integration assumptions prior to and independent of multi-core interactions. For a 128-core design, the core is the unit of replication; defects undetected at this layer replicate 128-fold across the die, amplifying area, power, yield, and debug cost. This document frames the research questions, surveys existing approaches (simulation, formal verification, hardware-assisted verification, FPGA prototyping, emulation), analyzes their applicability to a custom RISC-V core, and identifies the gaps that the XH-1 project must close given that the existing related documents in `research/10-verification/` are placeholders ("SOON"). + +## Research Question + +What verification methodology, toolchain, environment abstraction, coverage model, and sign-off criteria are appropriate for verifying a single XH-1 core such that the methodology: + +1. Provides evidence that the core implements the RISC-V ISA subset selected for XH-1 (see `research/10-verification/isa-compliance.md`). +2. Provides evidence that the core's microarchitecture meets timing, area, and power targets for a design that will be replicated 128 times. +3. Scales economically — verification effort, compute cost, and human review time must not grow linearly with the 128-core replication factor. +4. Produces artifacts (testbenches, coverage models, formal proofs, regression suites) that compose cleanly with the surrounding verification layers (unit, multicore, system, post-silicon). +5. Is implementable given that the related XH-1 documents in area 10 are not yet populated, indicating a greenfield verification program. + +## Background + +Verification of a processor core is conventionally decomposed into: + +- **Unit verification** — verification of individual blocks (ALU, FPU, register file, branch predictor, fetch/decode structures, load-store unit, replay logic, CSR file, MMU/TLB, caches). See `research/10-verification/unit-verification.md`. +- **Core verification** — verification of the integrated core as a single design under test (DUT), exercising instruction-stream and exception behavior at the architectural interface, plus internal microarchitectural invariants. +- **Multicore verification** — coherence, consistency, interconnect, and shared-resource correctness across multiple core instances. See `research/10-verification/multicore-verification.md`. + +For XH-1, the core is the replication unit. The cost model must distinguish two quantities that this document often refers to together as "cost": + +- **Defect-escape cost (multiplied by replication).** The downstream cost of an undetected defect (silicon debug, re-spin, field failure, yield loss). Under the replication-homogeneity assumption, an undetected core defect is replicated 128-fold in silicon, multiplying its downstream impact. +- **Verification effort cost (paid once).** The engineering cost of running the testbench, discharging formal properties, maintaining the regression, and writing tests. This is paid against one RTL; the fix, once made, is replicated. + +The phrase "every bug caught at the core layer is caught 128 times" is shorthand for the first quantity: every defect that escapes core verification multiplies its downstream cost by the replication count. It is not a claim that the verification effort itself runs 128 times — the testbench runs against the RTL once. + +**ASSUMPTION (replication homogeneity):** The 128 XH-1 cores are architecturally identical instances of a single RTL core produced by elaboration of the same source. Heterogeneous cores, per-core configuration, or runtime variants invalidate the replication-leverage argument on defect-escape cost and require per-variant verification. This is treated as an assumption because no repository evidence contradicts it; OPEN QUESTION 6 below revisits it. The "per-instance marginal cost of zero" claim for proven formal properties holds for bit-identical replicated RTL and refers to the marginal cost of *re-running* a proof against another elaborated instance of the same source — not to the engineering cost of constructing the proof. If the source is parameterized and elaborated with per-instance parameters that differ, the proof must be re-established for any parameter set that differs, and that re-establishment cost is not zero. + +**ASSUMPTION (target implementation class):** XH-1 targets ASIC tape-out. FPGA-only deployment would change the methodology (no DFT/ATPG, different sign-off criteria, no post-silicon bring-up in the same sense). This assumption is conditional and is revisited in OPEN QUESTION 12. + +**ASSUMPTION (verification language availability):** The verification language of the testbench is the same as or interoperable with the design language. This is revisited in OPEN QUESTION 11. + +Core verification draws on several formal and semi-formal techniques: + +- **Dynamic simulation** with constrained-random or directed tests, against an ISA reference model, with functional coverage. +- **Static (formal) verification** — model checking (bounded model checking, k-induction, IC3/PDR), equivalence checking (combinational and sequential), and property proving against the RTL. +- **Hardware-assisted verification** — FPGA prototyping and emulation (commercial emulators, open-source flows such as FireSim on Amazon EC2 F1), trading compile time for runtime speed. +- **Co-simulation** — two simulation engines running in lockstep (e.g., an RTL simulator and an architectural reference-model engine such as Spike) synchronized at defined events. This is distinct from **reference-model checking**, in which the reference model is integrated as a transactor within a single simulator; the latter is a common industrial pattern and is more accurately described as reference-model checking, not co-simulation in the strict sense. +- **Waveform and assertion-based methodologies** — SystemVerilog Assertions (SVA), PSL, custom monitors, end-of-test checkers. + +Coverage is a central concern. Functional coverage, code coverage (line, branch, expression, toggle, FSM), and cross-coverage of architectural state are the standard dimensions. For a custom ISA implementation, the relationship between ISA compliance coverage and microarchitectural coverage must be defined explicitly. + +Standard core-verification concerns that must be addressed in addition to instruction-stream behavior: + +- **Lint** — static analysis of RTL style, common bug patterns, and synthesizability (e.g., Verilator lint, commercial lint tools). A prerequisite to dynamic and formal work. +- **Clock-domain crossing (CDC) and reset-domain crossing (RDC)** — analysis of signals crossing asynchronous clock or reset domains. Typically owned by a separate methodology but the core must be CDC/RDC-clean for integration. +- **X-propagation** — controlled handling of unknown ('X) propagation in simulation, using X-pessimism removal or X-init optimization, to avoid false coverage from masked X's. +- **Reset verification** — power-on reset, software reset, debug reset, partial reset, and reset-domain interaction. +- **Power intent** — if power features are present (see OPEN QUESTION 7), UPF/CPF power intent formats integrate with simulation to model power-gated, retention, and isolation behavior. +- **DFT/ATPG interaction** — scan, BIST, and MBIST structures must not alter mission-mode behavior; their interaction with the core testbench is a coordination point. + +These are not optional; their absence in a verification plan is a known gap in pre-silicon methodology. + +## Existing Approaches + +The following approaches are established in industry and academic practice. Their applicability to XH-1 is analyzed; none is recommended in isolation. + +### A. Constrained-Random Simulation with UVM + +**FACT:** Universal Verification Methodology (UVM) is standardized as IEEE 1800.2-2017, *Standard for Universal Verification Methodology* [IEEE 1800.2-2017], and is widely used in industrial block- and core-level processor verification for constrained-random stimulus generation, reference model integration, self-checking scoreboards, functional coverage, and phasing. The Accellera UVM reference implementation, which is the source code that vendors and users actually consume, is distributed under the Apache License 2.0; the IEEE 1800.2-2017 standard document itself is an IEEE publication and is not freely redistributable. The two are distinct artifacts and this distinction is load-bearing when characterizing "open-source viability" below. + +**ASSUMPTION (conditional on design language):** UVM is applicable only if the XH-1 design language supports SystemVerilog class constructs — either the source is SystemVerilog, or the verification environment targets SystemVerilog regardless of source language. If the design language is Verilog-2001 only, or if the project commits to a Chisel/Spinal HDL or Bluespec-only flow, UVM is not directly applicable at the HDL-source level and an alternative testbench harness is required. This assumption is revisited in OPEN QUESTION 11. + +**PROPOSAL (applicability to XH-1):** Apply UVM as the primary dynamic verification environment for XH-1 core-level testing, paired with a chosen ISA reference (Spike or Sail). Sequence generation over the RISC-V ISA state space; reference model in co-simulation; scoreboard for architectural state comparison per committed instruction; explicit cover points for instruction × operand-mode × privilege × exception interactions. + +### B. Directed and Hand-Written Tests + +**FACT:** Constrained-random alone does not reach corner-case coverage of complex pipelines, corner-case CSR interactions, or rare exception sequences. Industry practice supplements with directed tests written by architects and designers. + +**PROPOSAL (applicability to XH-1):** A XH-1-specific directed test suite for pipeline hazard combinations specific to the XH-1 microarchitecture, corner cases in the chosen RVA profile subset, reset, wake, low-power entry/exit, and debug transport (if present at core level). + +### C. Formal Property Verification + +**FACT:** Bounded model checking, k-induction, and IC3/PDR are applied to control-dominated blocks: fetch, decode, CSR, MMU/TLB, branch predictor, retirement, and exception delivery. Commercial model checkers include Cadence JasperGold, Synopsys VC Formal, and Siemens FormalPro. Open-source flows include SymbiYosys (SBY), based on the Yosys synthesis and formal toolchain. Equivalence checking is a distinct formal category: tools include Synopsys Formality and Cadence Conformal, used to verify that two RTL representations (e.g., pre- and post-synthesis, or pre- and post-retiming) are functionally equivalent. Equivalence checkers are not model checkers and are not used for property proving. + +**PROPOSAL (applicability to XH-1):** Per-block formal sign-off for the control-dominated blocks listed above, end-to-end pipeline flushing, exception priority, and CSR read/write semantics. Formal proofs of privileged ISA semantics (e.g., `mstatus`, `sstatus`, `satp`, `mip`/`sip` behavior) provide a high-confidence anchor that random tests cannot match in isolation. Sequential equivalence checking between major RTL revisions is a separate, complementary activity. + +**ASSUMPTION (RISC-V privileged-ISA versioning):** The CSR semantics used as formal-proof targets correspond to the specific RISC-V Privileged ISA version adopted by XH-1. The unprivileged ISA, privileged ISA, and Debug specification each have versioned releases tracked by RISC-V International; the verification plan must be pinned to those versions. The specific version is OPEN until resolved (see OPEN QUESTION 1 and the Sources section). + +**PROPOSAL (applicability to XH-1):** Coordinate with `research/10-verification/formal-verification.md` for tool selection (commercial model checkers vs. SymbiYosys / SBY) and with `research/10-verification/isa-compliance.md` for the privileged-ISA version. + +### D. Reference-Model Checking and Co-Simulation + +**FACT:** Comparing DUT architectural state against Spike (the official RISC-V ISA simulator, C++ implementation) or Sail (a first-order imperative language designed for ISA specification, with a RISC-V model generated from the RISC-V ISA formal definition) per instruction is a widely deployed pattern. The terminology matters: + +- **Co-simulation (strict sense):** two independent simulation engines (e.g., an RTL simulator and Spike) running in lockstep, synchronized at defined events. The DUT may run ahead of the reference using a deferred-check queue, with results compared at synchronization points. +- **Reference-model checking:** a single simulator that incorporates the reference model as a transactor or foreign-language call. This is a common industrial pattern but is not co-simulation in the strict sense. + +**PROPOSAL (applicability to XH-1):** Required for ISA-compliance evidence. See `research/10-verification/isa-compliance.md`. The specific architecture proposed is **reference-model checking within a single simulator** (Spike invoked as a transactor from the UVM or lightweight harness) as the default for day-to-day runs, with **co-simulation in the strict sense** retained as a cross-check on a sampled subset of tests to detect synchronization artifacts that a single-simulator harness could mask. Reference model parallelism is a 128-core scaling concern: Spike and Sail are single-threaded by default; running 128 reference models in lockstep with 128 RTL instances is itself a significant engineering problem. The pattern of choice is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling (multicore-layer decision) with the core testbench supporting the interface. + +### E. FPGA Prototyping and Emulation + +**FACT:** FPGA-based prototyping runs the RTL at higher throughput than software simulation, and emulation platforms (commercial: Palladium, Veloce, ZeBu; open-source: FireSim on Amazon EC2 F1) provide a middle ground. The throughput multiplier relative to software simulation is workload- and instrumentation-dependent. + +**PROPOSAL (applicability to XH-1):** Use FPGA prototyping to speed up core testbench bring-up by running entire test programs end-to-end, to enable fault-injection campaigns that are infeasible in simulation (see `research/10-verification/fault-injection.md`), and as the natural place to begin validating the multicore layer; core verification is the upstream prerequisite. Commercial emulation is a viable alternative for very large workloads or when observability requirements are relaxed; the decision depends on licensing posture (OPEN QUESTION 3) and workload characteristics. No specific throughput multiplier for the XH-1 DUT is asserted here; INSUFFICIENT EVIDENCE without XH-1-specific RTL and platform data. + +### F. Regression Infrastructure + +**FACT:** Regression management — version control of testbenches, seeds, baselines, coverage databases — is a load-bearing component of any non-trivial core verification program. See `research/10-verification/regression.md`. + +**PROPOSAL (applicability to XH-1):** The XH-1 regression infrastructure must support a single-core nightly run, a multicore run, and a system run with shared seeds where appropriate. + +### G. Coverage Closure Methodology + +**FACT:** Coverage closure combines functional coverage (defined by the verification plan), code coverage (uncovered RTL), and assertion coverage. Industry practice treats coverage closure as a managed funnel: cover points → cross coverage → corner case identification → directed/random test addition → formal where possible. Coverage baselines drift as RTL changes; a baseline-management policy is required. + +**PROPOSAL (applicability to XH-1):** Codify the funnel explicitly in the verification plan with tracked stages, baseline-management policy, and exit criteria per stage. + +### H. Assertion-Based Verification + +**FACT:** SystemVerilog Assertions (IEEE 1800-2017, *Standard for SystemVerilog*) and immediate assertions encode design intent and serve both as live checkers and as inputs to formal tools. + +**PROPOSAL (applicability to XH-1):** Define an SVA library for XH-1 invariants (e.g., "no retired instruction has a privileged side effect inconsistent with its mode", "no speculative architectural state is observable at the architectural interface"). + +### I. Open-Source ISA-Level Formal Verification + +**FACT:** The `riscv-formal` framework (YosysHQ / Symbiotic EDA, Verilog, BSD-licensed) provides a checkable RISC-V ISA specification encoded as a set of formal properties that are conjoined with a wrapper around the DUT and discharged by an SMT solver. The Kami project (Cambridge) provides a formal RISC-V ISA specification in Coq with a Bluespec front-end. INSUFFICIENT EVIDENCE to assert at the FACT level that `riscv-formal` is "directly applicable" to a specific XH-1 microarchitecture and ISA profile without further evaluation: applicability depends on the DUT's memory and CSR interface matching the framework's wrapper assumptions, on the chosen RISC-V profile, and on the unprivileged/privileged ISA versions adopted. + +**PROPOSAL (applicability to XH-1):** Evaluate `riscv-formal` as a complement to block-level formal. It targets ISA-compliance evidence with open-source tooling and is particularly relevant given the greenfield status of the XH-1 verification program, subject to the applicability evaluation above. + +### J. Open-Source Constrained-Random Generator + +**FACT:** Google's `riscv-dv` is an open-source UVM-based constrained-random instruction-stream generator for RISC-V, widely used in industrial and academic flows. The license should be confirmed against the current repository state at the time of toolchain selection; this document does not assert a specific license version. + +**PROPOSAL (applicability to XH-1):** Evaluate `riscv-dv` as a generator either within a UVM environment or as a stimulus source for a lightweight harness. + +### K. RISC-V Architectural Test Framework + +**FACT:** RISC-V International maintains `riscv-arch-test` as the architectural test framework, which is the conformance test suite against which cores are signed off for profile compliance. `riscv-tests` is a related umbrella repository containing `riscv-arch-test` as well as additional tests; the relationship is not "legacy vs. replacement" — `riscv-tests` is the broader collection and `riscv-arch-test` is the architectural conformance subset. The two should be treated as distinct but related inputs, both of which produce evidence in the sign-off record. The framework's coverage analyzer, `RISCV-ISAC`, is an instruction-stream coverage analyzer used to confirm that the architectural test signatures cover the required instruction/operand/privilege combinations; it is not a generic functional-coverage database and is not a substitute for the core testbench's own coverage model. + +**PROPOSAL (applicability to XH-1):** Use `riscv-arch-test` for the ISA-compliance certificate, with `RISCV-ISAC` analysis of the architectural test signatures as part of the coverage evidence, and `riscv-tests` more broadly as a regression input. The license and current revision of each must be confirmed at toolchain selection. + +## Alternative Designs + +The following alternative *strategies* for the core verification methodology itself are candidates for XH-1. They are not mutually exclusive; the project must select a composition. + +### Alternative 1: UVM-Centric with Formal Supplements + +- **Approach:** UVM as the workhorse for dynamic verification, with formal applied to control-dominated blocks. +- **Strengths:** Industry-standard methodology, well-understood coverage and sign-off conventions, mature ecosystem; integrates with `riscv-dv` as a generator and `riscv-formal` as an ISA-level formal layer. +- **Weaknesses:** UVM has a steep learning curve and high setup cost. For a 128-core design, the UVM environment must be designed for reuse to avoid 128× rewrite cost. Requires SystemVerilog (OPEN QUESTION 11). Implies commercial simulator licensing in most deployments, because open-source SystemVerilog simulation support (Verilator) has limited UVM-class support relative to commercial offerings; the open-source viability of this alternative is therefore medium, not high. + +### Alternative 2: Lightweight Python/C++ Test Harness with Reference-Model Checking + +- **Approach:** A generator produces instruction streams and a C++ testbench (e.g., based on cocotb, Verilator, or a custom framework) compares architectural state against Spike. +- **Strengths:** Lower setup cost than UVM. Excellent generator productivity. Easy integration with regression infrastructure. Open-source-friendly. `riscv-dv` can serve as the generator. Verilator-based flows are not bound to SystemVerilog UVM at the source level. +- **Weaknesses:** Verilator's debug, SVA, and functional-coverage support are not equivalent to commercial simulators; cocotb+Verilator is therefore not a drop-in replacement for UVM, and the trade-off should be evaluated against the XH-1 debug and coverage requirements. Coverage closure tooling is less standardized than the UVM vendor ecosystem. The reference-model-checking pattern (Section D) is the natural fit for this harness. +- **Application to XH-1:** Reasonable choice if XH-1 has a small team, an open-source toolchain preference, or a constraint on commercial EDA licensing. + +### Alternative 3: Formal-First + +- **Approach:** Formal proofs form the primary verification evidence; dynamic simulation is used for reach and to validate properties that formal cannot discharge. +- **Strengths:** Highest assurance per proven property. Strong fit for control logic and CSR behavior. `riscv-formal` provides an open-source ISA-level formal layer subject to the applicability evaluation in Section I. Under the replication-homogeneity assumption, proven properties transfer to all 128 instances at near-zero marginal *re-proof* cost (the proof is re-run, not re-constructed, against another elaborated instance of the same RTL); the engineering cost of constructing the proof is paid once. +- **Weaknesses:** Pipeline formalization is hard. State-space complexity grows combinatorially for large cores. May not reach proof closure on timing-agnostic models without abstraction. Inconclusive results are not equivalent to failure; risk-acceptance must be tracked. + +### Alternative 4: Emulation-Centric + +- **Approach:** Skip large-scale software simulation; use FPGA/emulation throughput as the primary engine. +- **Strengths:** Excellent long-workload capability; native fault-injection support. +- **Weaknesses:** Poor waveform/observability; slow compile/turnaround; harder to drive corner-case stimulus; debug cycle is slow. + +### Alternative 5: Hybrid (PROPOSAL — See Proposed Direction) + +A combination of a dynamic framework (Alternative 1 or Alternative 2) for stimulus and reach, formal supplements (Alternative 3, including `riscv-formal` where applicable) for control logic, and FPGA prototyping/emulation for throughput and multicore bring-up. + +## Comparison + +The comparison is qualitative and is intended to make the trade-offs explicit, not to produce a numeric ranking. INSUFFICIENT EVIDENCE exists to rank absolute quality without XH-1-specific RTL data, tool licensing posture, and team size. The Hybrid column is a composition of the others and is not an independent point of comparison; the table is structured to show how each pure strategy performs on the dimensions, with the Hybrid column indicating how the strengths compose. + +| Dimension | UVM-Centric | Lightweight Harness | Formal-First | Emulation-Centric | Hybrid | +|---|---|---|---|---|---| +| Setup cost | High | Low | High | Medium | High | +| Per-test runtime | Slow | Slow–Medium | N/A (proof time) | Fast | Mixed | +| Corner-case reach | Strong (with seeds) | Strong (with fuzzing) | Strong (when discharged) | Weak–Medium | Strong | +| Assurance on control logic | Medium | Medium | High | Low–Medium | High | +| Debug ergonomics | Good (waveform, logs) | Good (caveat: Verilator/cocotb debug support not equivalent to commercial) | Strong counterexamples | Weak | Mixed | +| Open-source viability | Medium (UVM standard is IEEE-published; UVM reference implementation is Apache 2.0; simulators are usually commercial) | Strong | Medium | Medium | Medium | +| Fit for 128-core reuse | High (if designed so) | High (if designed so) | High (under replication homogeneity) | High | High | +| Pre-tape-out risk reduction | Strong | Medium–Strong | Strong (partial; inconclusive ≠ proven) | Medium | Strong | + +## Advantages + +The advantages of investing in a robust core verification methodology for XH-1: + +- **Replication leverage on defect-escape cost.** Under the replication-homogeneity assumption, an unfixed defect in the core RTL is replicated 128-fold in silicon, multiplying its downstream (silicon-debug, re-spin, field-failure, yield-loss) cost. Verification effort itself is paid once against one RTL; the defect, if found, is fixed once; the fixed RTL is then replicated. The benefit of core-level detection is the avoidance of multiplied post-silicon cost. +- **Cleaner multicore verification.** A well-verified core reduces the multicore verification problem to coherence, consistency, interconnect, and shared resources. See `research/10-verification/multicore-verification.md`. +- **Foundation for post-silicon.** First-stage silicon bring-up is dramatically simpler when the core is well-understood and the debug interfaces (assertions, traces, scan) are well-designed. Post-silicon validation feeds back into pre-silicon coverage targets; this feedback loop is not detailed in this document and is left to the post-silicon layer. +- **Reusable testbench assets.** Generator, scoreboard, coverage model, and SVA library can be reused across cores if XH-1 has core variants, or across revisions of the same core. +- **Formal evidence on critical control.** Privileged ISA semantics are hard to verify by random simulation; formal proofs on these blocks provide a high-confidence floor. + +## Disadvantages + +Costs and risks of a robust core verification program: + +- **Engineering cost.** A UVM-class environment plus formal plus emulation is a substantial program. The claim that it "can easily exceed the RTL design cost for a single core" is folkloric and is presented here as a planning consideration, not as a measured fact. +- **Tooling lock-in.** Heavy UVM usage often implies commercial simulator licensing; this may not match XH-1's tool posture. Formal tool lock-in is similar: commercial model checkers differ in property language and run scripts. +- **Slow turnaround.** Constrained-random simulation with full coverage at the core level is slow; coverage closure can take many cycles of "add test, rerun, close hole". Coverage baselines drift with RTL changes, requiring re-establishment. +- **Reference model dependence.** Spike, Sail, or any reference may diverge from the RISC-V specification in interpretation of ambiguous or profile-specific corners; the verification program must track the chosen RVA profile carefully and pin the spec version of the reference. +- **Coverage false confidence.** Hitting a cover point is not the same as proving correctness; the verification plan must distinguish cover and check. +- **128-core replication pressure.** If the core testbench is not designed to compose with the multicore testbench, redundant effort emerges. +- **Inconclusive formal ≠ proven.** Formal results of "inconclusive" must be risk-accepted, not treated as success. +- **Equivalent mutants.** Mutation-kill rates are affected by equivalent mutants (mutations that produce behaviorally identical RTL); kill-rate targets are therefore noisy and must be interpreted with caution. + +## XH-1 Considerations + +XH-1-specific factors that constrain the methodology choice: + +- **128-core replication.** The verification plan must be replication-aware: the *testbench* should be agnostic to whether one or 128 cores are instantiated. This implies strong separation of (a) the core DUT, (b) the core testbench, and (c) the multicore wrapper. +- **Custom design.** No third-party core is being verified; there is no vendor's pre-built verification IP. All environments must be built or assembled. +- **Repository context.** The related verification documents are all placeholders ("SOON"). This is a greenfield verification program; the XH-1 project has the opportunity — and the burden — to define the methodology from first principles. +- **RISC-V ISA.** The RISC-V ecosystem provides mature reference models (Spike, Sail), compliance / architectural test frameworks (`riscv-arch-test`, `riscv-tests`), and open-source verification tooling (`riscv-formal`, `riscv-dv`, `RISCV-ISAC`). These are the natural anchors. +- **Tool posture (OPEN QUESTION 3).** Without evidence in the repository, the project's tool posture (open-source only, commercial EDA, mixed) is OPEN. The methodology composition must remain valid across posture choices. +- **Design language (OPEN QUESTION 11).** The choice of design language — SystemVerilog, traditional Verilog, Chisel/Spinal HDL, Bluespec — has direct implications for verification methodology (e.g., formal property availability on generated RTL, simulation throughput, the place of unit-level testing at the HDL-source level vs. the generated-RTL level, and the applicability of UVM at the source level). This is OPEN pending XH-1 design-language disclosure. Section A's UVM assumption is conditional on this resolution. +- **Microarchitectural class (OPEN QUESTION 2).** The pipeline class — in-order vs. out-of-order, superscalar width, depth of speculation, branch-prediction aggressiveness — materially changes the cost-benefit of formal vs. dynamic and the complexity of pipeline flush / replay verification. The methodology composition must accommodate both extremes. +- **Power architecture (OPEN QUESTION 7).** Whether the core supports clock gating, power gating, retention, and/or DVFS is OPEN. The verification methodology must accommodate each combination. Assuming any specific power feature is invalid without evidence. Power intent (UPF/CPF) integration is conditional on this. +- **Target implementation class (OPEN QUESTION 12).** Whether XH-1 targets ASIC tape-out or FPGA-only deployment fundamentally changes the verification methodology (DFT/ATPG scope, post-silicon layer, sign-off criteria). Treated here as an assumption (ASIC) pending resolution. + +## 128-Core Scalability + +Replication does not change which verification techniques apply to a single core, but it changes their cost-benefit and their composition. Throughout this section, the two cost quantities from the Background are kept distinct: **defect-escape cost** (multiplied by replication count under the homogeneity assumption) versus **verification effort cost** (paid against one RTL). + +- **Core testbench scalability.** A core testbench written for a single instance must be instantiable inside a multicore wrapper without modification. This implies the testbench interfaces to the core through a standard transaction-level interface (or at least a clearly defined boundary), not through global signals. +- **Cost-benefit of formal at replication.** Per-block formal verification has high fixed engineering cost (constructing the proof), but under bit-identical replication, the *re-proof* cost against another elaborated instance of the same RTL is near zero: once a control property is proven, re-running the proof against a re-elaborated instance of the same source is mechanical. This is a property of the proof, not of the underlying effort. If the source is parameterized and elaborated with per-instance parameters that differ, the proof must be re-established for any parameter set that differs. The defect-escape cost argument is the dominant benefit: a proven property that prevents a class of bug multiplies that benefit by the replication count. +- **Cost-benefit of dynamic at replication.** Dynamic simulation at full coverage also has its verification effort paid against one RTL, but full per-instance coverage at 128 instances requires running 128 instances through the same stimulus, which is a real additional compute cost (not strictly linear in wall-clock because the simulator elaborates them in the same run, but real in host compute and memory). The trade-off is a primary argument for the hybrid approach at 128-core scale, and the practical mitigation is sampling, not full per-instance coverage. +- **Workload scaling.** Running the full core testbench in a 128-core environment is generally unnecessary; sampling (verify a subset of cores under full stimulus, sweep register state across all 128) is a more efficient pattern. The sampling strategy is a multicore concern but must be supported by the core testbench's reset/configuration interface. The statistical argument for the chosen sampling fraction (defect-escape probability vs. simulation cost) is a multicore-layer deliverable; the core testbench must support the interface. +- **Coverage aggregation.** Per-core coverage data must be aggregated across the 128 instances. The core testbench should emit coverage in a form that is identifiable by instance index. +- **Debug bandwidth.** 128 cores each generating waveform/log data is impractical. The core testbench must distinguish between detailed-trace mode (one core, full visibility) and lightweight-trace mode (128 cores, summary only). +- **Compile and elaboration cost.** Elaboration of a 128-core testbench can dominate runtime. The core testbench should be elaborable standalone and in a multicore wrapper without recompiling the core RTL. Elaboration-time behavior (reset, clock tree, memory subsystem) can differ between 1-core and 128-core contexts; the core testbench must be valid in both. +- **Testbench memory footprint.** 128 instances of a complex scoreboard and 128 reference models can dominate host memory. Memory-aware testbench design (shared reference models, deferred-check queues, summary-only trace mode) is required. This is acknowledged here and detailed in the multicore layer. +- **Reference-model parallelism.** Spike and Sail are single-threaded by default. Running 128 reference models in lockstep with 128 RTL instances is a significant engineering problem. The recommended pattern is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling. This is a multicore-layer design constraint that the core testbench interface must support. +- **Symmetry assumption revisit.** The replication-homogeneity assumption is load-bearing throughout this section. Heterogeneous cores (e.g., big/little, mixed application/accelerator, per-core ISA-subset configuration) invalidate the per-die marginal-cost argument and require per-variant verification, including per-variant testbenches, per-variant coverage, and per-variant formal. The methodology should be designed so that the additional cost of a variant is linear in the number of variants, not linear in 128 × variants. + +## Performance Considerations + +The following are framework-level considerations. INSUFFICIENT EVIDENCE exists to provide numerical throughput estimates without XH-1-specific RTL data, simulator choice, and toolchain selection; any specific number is unsourced and is therefore not asserted here. + +- **Testbench throughput.** Dynamic simulation throughput determines how many random seeds can be run per unit wall-clock. Constrained-random at the core level is sensitive to simulator choice, RTL abstraction level, and coverage instrumentation. ASSUMPTION: representative numbers are available in vendor documentation (commercial simulators) and in the open-source Verilator and cocotb benchmark literature, but are workload-dependent and cannot be quoted as typical without measurement on the XH-1 DUT. No specific throughput range is asserted. +- **Compiler/optimizer.** Coverage instrumentation slows simulation. Production coverage runs and fast-turnaround debug runs should be separable. +- **Co-simulation overhead.** PROPOSAL: Spike co-simulation overhead per committed instruction is non-trivial in co-simulation in the strict sense (two-engine lockstep); under the reference-model-checking pattern adopted in Section D, the overhead is a foreign-language call per synchronization point. Synchronization strategy (per-instruction vs. batched) is a tradeoff between debug granularity and throughput. At 128 instances, the cost of a per-instruction synchronization across all instances is prohibitive; batched or sampled synchronization is the practical pattern. +- **FPGA prototype throughput.** FPGA prototype throughput is set by Fmax, which is determined by LUT count, routing, and the memory subsystem. Quoted MHz figures for an XH-1 core on a specific FPGA platform are not asserted here; INSUFFICIENT EVIDENCE without platform and synthesis data. +- **Regression wall-clock budget.** A nightly regression that runs N seeds on a single core must fit within the available compute budget. Cloud burst capacity is typical. + +INSUFFICIENT EVIDENCE to provide numerical throughput estimates without XH-1-specific RTL data and toolchain selection. + +## Area Considerations + +- **On-die verification infrastructure.** XH-1 is described as a 128-core design. The verification *methodology* does not directly consume die area, but decisions made during verification (e.g., embedding assertions in RTL, adding observability muxes for trace, adding per-core DFT structures) do. +- **Trace and debug infrastructure.** Core-level verification informs the design of trace, performance counters, and debug-transport blocks. Larger trace infrastructure costs area; the verification program must justify the cost in terms of debug productivity and post-silicon coverage. +- **DFT (Design-for-Test) compatibility.** Scan chains, BIST, and MBIST are typically owned by a separate DFT program. Core verification should ensure that the core's DFT structures do not alter functional behavior in mission mode. If XH-1 is FPGA-only (OPEN QUESTION 12), DFT is out of scope. +- **Replication-driven area optimization.** If the core's area is a major constraint (which is likely at 128 cores on a single die), the verification plan must protect against defects that would force a re-spin with re-spinned core area. + +INSUFFICIENT EVIDENCE to quantify area without access to XH-1's core area budget or process node. + +## Power and Energy Considerations + +The following considerations apply **if and only if** XH-1 implements the corresponding power features. Each is OPEN QUESTION 7-conditional. + +- **Power-aware verification (conditional).** If the core supports power gating or retention, the core testbench must be able to drive representative switching activity to enable power estimation, and the testbench must verify that power-gated domains behave correctly (controlled wake, retention restore, ordering with respect to coherence traffic). UPF/CPF power intent integration with the simulator is the standard mechanism. +- **Energy proportionality (conditional).** If the core supports clock gating at the instruction or block level, the testbench should verify that idle and active transitions behave correctly at the architectural interface (no observable state loss in idle, correct wake on interrupt). +- **Thermal and DVFS interactions (conditional).** If XH-1 supports DVFS at the core level, voltage/frequency transitions and their effects on memory coherence, pipeline flush behavior, and CSR access must be verified. This is a cross-cutting concern between core and system verification. +- **Power estimation accuracy.** Pre-silicon power estimation depends on workload realism. A core testbench that only runs synthetic random code will produce misleading power numbers. + +No power feature is assumed. Verification provisions for any power feature must be activated only when OPEN QUESTION 7 is resolved. + +## Implementation Considerations + +- **Methodology composition.** The project must select one of the alternative strategies (or a hybrid) and codify it in the verification plan. +- **Toolchain selection.** Open-source (Verilator, cocotb, SymbiYosys, `riscv-formal`, `riscv-dv`, Spike, RISCV-ISAC, GTKWave) vs. commercial (Synopsys VCS, Cadence Xcelium, Siemens Questa, JasperGold, VC Formal, FormalPro, Formality, Conformal, Palladium, Veloce, ZeBu) is a strategic decision. The choice cascades through licensing, hiring, training, and CI. These categories are not equivalent in role: Verilator and cocotb are simulation/cosim frameworks; SymbiYosys is a formal-verification flow; `riscv-formal` is an ISA-level formal property set; `riscv-dv` is a stimulus generator; RISCV-ISAC is an instruction-stream coverage analyzer for the architectural test framework; commercial emulators are hardware-assisted verification platforms distinct from FPGA prototyping. The choice within each role is independent. +- **Reference model choice.** Spike (C++, instruction-accurate) vs. Sail (first-order imperative ISA-specification language with a RISC-V model) vs. a custom golden model. Spike is the most common reference; the project must document any corner-case behaviors of the chosen reference against the RISC-V specification. +- **Coverage database.** Coverage data are stored in vendor-specific formats (e.g., Synopsys UCDB / DDC, Cadence IMC, Siemens UCDB) and may be aggregated into a custom backend (e.g., SQLite) for CI. The format is a tool-vendor concern; the aggregation backend is a CI concern. These are distinct choices. +- **Test categorization.** Tests must be tagged by domain (base integer, M-ext, A-ext, F/D-ext, privileged, vector, …), by intent (sanity, corner, regression, performance), and by status (passing, known-fail, expected-fail). +- **Seed management.** Random seeds should be reproducible. Seed pools should be partitioned into "core regressions", "block regressions", and "fuzzing". +- **Bring-up sequencing.** Block-level → core-level → multicore → system. The core testbench is the transition point between block-level and system-level work. + +## Sign-Off Criteria + +A core verification sign-off is the point at which the methodology exits the coverage-closure funnel. The following are candidate criteria; **every numeric threshold below is an illustrative starting point and is not a validated target**. The relationship to DRAFT status is restated at each threshold so that the list is internally consistent with the document status header. + +- **Functional coverage.** All defined cover points hit, with cross-coverage exercised. **Illustrative starting threshold (not a validated target):** ≥ 95% cover-point hit, ≥ 90% cross-coverage hit; remaining items must be justified as unreachable or formally proven. Final threshold to be set in the verification plan. +- **Code coverage.** Statement, branch, expression, toggle, and FSM coverage reported; excluded regions justified. **Illustrative starting threshold (not a validated target):** ≥ 95% statement, ≥ 90% branch, with manual exclusion of reset/DFT/clock-gating logic. Final threshold to be set in the verification plan. +- **Assertion coverage.** All SVA / immediate assertions vacuously satisfied or actively proven; assertion cover points hit. Open assertions must be reviewed and dispositioned. +- **Formal closure.** Per-block formal sign-off, with proof closure recorded for each property. **Operational requirement:** every inconclusive formal result must have a documented risk-acceptance disposition (strengthen via abstraction/decomposition, or accept with justification) recorded in the sign-off package. "Inconclusive" is not equivalent to "proven". +- **ISA compliance evidence.** All applicable tests in `riscv-arch-test` pass and produce the architectural test framework's signature. `riscv-tests` (broader collection including `riscv-arch-test`) is run as a regression input. Signed certificate or equivalent record retained. **Prerequisite:** the specific RISC-V ISA, Privileged ISA, Debug, and Profile versions adopted by XH-1 are pinned in the verification plan. +- **Mutation testing.** Periodic injection of known-defect mutations into a frozen RTL revision. **Illustrative starting threshold (not a validated target):** ≥ 80% of injected mutations detected by the testbench + assertions + formal. This number is a starting point and requires empirical validation on the XH-1 DUT. Equivalent-mutant handling (mutations that produce behaviorally identical RTL) must be defined; kill-rate targets are noisy and must be interpreted with caution. Final threshold and operator set to be set in the verification plan. +- **Reference model validation.** Spike or Sail validated against the chosen RISC-V compliance tests before being used as a checker; validation evidence archived. +- **Testbench self-check verification.** Every scoreboard check has at least one negative test that is expected to fail. +- **Reproducibility.** Every sign-off run is reproducible from committed source, seed, and tool version. +- **CDC/RDC closure.** CDC and RDC analysis clean for all core-level crossing structures. +- **Lint clean.** Lint clean at the policy-defined severity level. +- **Reset verification.** All reset domains and reset sequences verified; reset recovery verified. + +## Verification Considerations + +Core verification is itself subject to verification — the verification of the verification environment: + +- **Testbench self-checking.** All scoreboard checks must be end-of-test verifiable; every check should be paired with at least one negative test that is expected to fail. +- **Reference model validation.** The reference model (Spike, Sail) must be validated against the chosen RISC-V compliance tests before being used as a checker. +- **Coverage sanity.** Cover points must be reviewed to ensure they are not trivially self-firing. +- **Mutation testing.** A subset of RTL defects should be intentionally injected to confirm the testbench catches them. See `research/10-verification/fault-injection.md`. Equivalent-mutant handling must be defined. +- **Coverage baseline stability.** Coverage baselines drift as RTL changes; a baseline-management policy (when a baseline is re-established, by whom, with what approval) is required. +- **Tool qualification.** If the verification tools are used in any safety-relevant context, tool qualification (per ISO 26262, IEC 61508, DO-254, or similar) is required. The applicability of these standards to XH-1 is OPEN QUESTION 8. Tool qualification is a verification-of-verification concern, not only a sign-off checkbox. +- **Version control and baselines.** Every test run must be reproducible from committed source. Coverage baselines must be versioned. +- **Testbench verification at 128-instance scale.** The testbench itself has bugs. Verification of the testbench (negative tests, mutation of the testbench, equivalence of testbench revisions) is part of the methodology. At 128 instances, the testbench's own correctness becomes a non-trivial verification problem and is acknowledged here as a multicore-layer concern that the core testbench interface must support. + +## Software Considerations + +- **Toolchain integration.** The core verification environment must integrate with the software toolchain (compiler, assembler, linker, Spike, debug server) to enable reference-model checking / co-simulation and to run real programs. +- **Firmware.** Firmware that runs on the core (boot code, exception handlers, interrupt dispatchers) is itself verification target; the core testbench should support loading and running firmware images. +- **Operating system implications.** A core that supports an OS (Linux, RTOS) must be verified with syscall-heavy workloads. The core testbench should provide OS-aware run modes. +- **Debug and trace protocols.** The core's debug-transport module (per the RISC-V Debug specification) must be exercised. JTAG or cJTAG interactions with the core's debug module are typically a core-level concern. +- **Performance analysis tools.** Tools that analyze trace data, performance counters, and profile information should be able to consume the artifacts the core testbench produces. +- **Software developer ergonomics.** The verification team's interfaces (test writers, coverage reviewers) should be ergonomic. A "developer can run a single test and see results in under a minute" goal is a useful operational target. + +## Security Considerations (Out of Scope, but Flagged) + +**ASSUMPTION (out-of-scope for this revision):** Transient-execution / side-channel verification (e.g., Spectre-class, Meltdown-class, and constant-time verification) is treated as out of scope for the XH-1 core verification layer for this revision. For a 128-core design with privilege boundaries, this is a real and high-risk concern. This is flagged as a high-risk gap and as a candidate for a follow-up document. The current methodology is sufficient for functional and microarchitectural correctness but does not, by itself, provide security assurance. + +## Proposed Direction + +PROPOSAL — A **hybrid methodology** for XH-1 core verification, with the following structure: + +1. **Dynamic simulation** as the primary verification engine, with the framework (UVM vs. lightweight) gated on OPEN QUESTION 3 (tool posture), OPEN QUESTION 4 (team size and timeline), and the design-language choice (OPEN QUESTION 11). + - If UVM: pair with `riscv-dv` as the instruction-stream generator and Spike or Sail as the reference model. + - If lightweight: pair `riscv-dv` (or a custom generator) with cocotb / Verilator and Spike, with explicit acknowledgment of Verilator's reduced debug, SVA, and coverage support relative to commercial simulators. +2. **Formal verification** applied to control-dominated blocks: fetch, decode, CSR file, MMU/TLB, exception/interrupt delivery, and the debug module. Coordinate with `research/10-verification/formal-verification.md` for tool selection. Apply `riscv-formal` as an ISA-level formal layer subject to the applicability evaluation in Section I. Sequential equivalence checking between major RTL revisions is a complementary activity. +3. **Reference-model checking against a RISC-V reference model** (Spike or Sail) as the default, with **co-simulation in the strict sense** retained as a sampled cross-check (per Section D). Coordinate with `research/10-verification/isa-compliance.md`. Pin the RISC-V ISA, Privileged ISA, Debug, and Profile versions. At 128 instances, use a shared reference model with a deferred-check queue or sampled synchronization. +4. **Directed tests** for XH-1 microarchitectural corner cases that random cannot reach economically. The corner cases to target are conditional on the in-order / out-of-order answer in OPEN QUESTION 2. +5. **FPGA prototyping** for throughput-driven verification and as a substrate for fault injection and multicore bring-up. Commercial emulation is a conditional alternative based on licensing posture (OPEN QUESTION 3) and workload. +6. **Regression infrastructure** with version-controlled seeds, baselines, and coverage data. Coordinate with `research/10-verification/regression.md`. +7. **Coverage closure** as an explicit, tracked funnel with the sign-off criteria in the Sign-Off Criteria section, including baseline-management policy. Numeric thresholds are illustrative starting points and are not validated targets; final thresholds are deferred to the verification plan. +8. **Reusable testbench architecture** so the core testbench composes inside the multicore testbench without modification, supporting both the per-die cost argument and the sampling strategy. The interface must support shared reference models, summary-only trace mode, and per-instance coverage identification. +9. **Mutation testing** as a periodic check that the testbench is still doing useful work, with explicit equivalent-mutant handling. Coordinate with `research/10-verification/fault-injection.md`. +10. **Standard core-verification adjuncts:** lint, CDC/RDC analysis, X-propagation handling, reset verification, and (conditional on OPEN QUESTION 7) UPF/CPF power-intent integration. + +The hybrid is proposed; the specific framework within the hybrid (UVM vs. lightweight) is gated on OPEN QUESTION 3, OPEN QUESTION 4, and OPEN QUESTION 11, and the power-feature sub-portion is gated on OPEN QUESTION 7. The implementation-class sub-portion (DFT/ATPG scope, post-silicon layer) is gated on OPEN QUESTION 12. The document does not pick between alternatives that the OPEN QUESTIONS do not yet support. + +## Confidence + +- **High confidence:** The general methodology composition (dynamic + formal + reference-model checking / co-simulation + directed + prototype + regression + coverage closure, with the standard adjuncts: lint, CDC/RDC, X-propagation, reset) corresponds to the IEEE 1800.2-2017 UVM standard, the IEEE 1800-2017 SystemVerilog assertion subset, and the RISC-V International published ISA / Privileged / Debug / Profile specifications, and is broadly accepted in industrial practice. The references supporting each component are cited at the point of use. +- **High confidence (conditional):** The 128-core scaling argument on defect-escape cost (an undetected defect is replicated 128-fold in silicon) is a direct consequence of the replication-homogeneity assumption and does not require empirical data. The argument about formal re-proof cost under bit-identical replication (near-zero marginal cost to re-run a proof against another elaborated instance of the same source; engineering cost of constructing the proof is paid once) is also a direct consequence of the assumption, with the parameter-set caveat acknowledged. The argument about dynamic simulation at full per-instance coverage requiring real additional compute is similarly structural. +- **Medium confidence:** Specific tool selections, sign-off numeric thresholds, and area/power tradeoffs, given the absence of repository evidence on the OPEN QUESTIONS. +- **Low confidence (OPEN):** XH-1's tool licensing posture, core microarchitecture details, ISA profile selection, team size, design language, power architecture, and target implementation class. These drive the specifics within the proposed hybrid. + +## Open Questions + +1. What RISC-V profile does the XH-1 core implement, and what is the specific version pin for the Unprivileged ISA, Privileged ISA, and Debug specification? The reference model (Spike variant, Sail configuration) and the compliance test scope follow from this choice. The RISC-V International profile document should be consulted at toolchain selection to determine the exact profile designation adopted (e.g., a 64-bit XLEN core targeting an RVA application/supervisor profile, or a 32-bit XLEN core targeting a different profile track); this document does not assert a specific profile name as established fact. +2. Is the XH-1 core in-order or out-of-order? The verification plan complexity differs substantially. The instructions explicitly disallow assuming either without repository evidence. +3. What is XH-1's toolchain posture (open-source only, commercial EDA, mixed)? +4. What is the team size and timeline? Methodology cost scales with team capacity. +5. What is the core's microarchitectural contract — what blocks are in scope for unit verification, and what is the boundary between core and multicore? See `research/10-verification/unit-verification.md` and `research/10-verification/multicore-verification.md`. +6. Are the 128 cores architecturally identical, or are there variants (big/little, mixed application/accelerator, per-core ISA-subset configuration)? Heterogeneity changes the testbench architecture and invalidates the per-die cost-leverage argument. +7. What is the core's power architecture — is clock gating, power gating, retention, or DVFS at the core level, and what are the state-retention semantics? +8. What is XH-1's relationship to functional safety standards (ISO 26262, IEC 61508, DO-254, or none)? +9. How will the XH-1 core's trace, debug, and performance-counter interfaces be designed, and how will core verification consume them? +10. What is the expected defect rate target at tape-out, and how does the verification program demonstrate closure against that target? +11. What is the XH-1 design language (SystemVerilog, Verilog-2001, Chisel/Spinal HDL, Bluespec), and does verification target the HDL source level, the generated-RTL level, or both? +12. What is XH-1's target implementation class — ASIC tape-out, FPGA-only deployment, or both? This determines DFT/ATPG scope, post-silicon validation depth, and sign-off criteria. +13. What physical implementation tool flow (synthesis, PnR, timing) is envisioned, and what are the timing-closure constraints (target clock frequency, slack budget)? + +## Sources + +- IEEE 1800.2-2017, *Standard for Universal Verification Methodology*. Defines UVM. The Accellera UVM reference implementation is distributed under the Apache License 2.0; the IEEE 1800.2-2017 standard document is an IEEE publication. +- IEEE 1800-2017, *Standard for SystemVerilog — Unified Hardware Design, Specification, and Verification Language*. Defines SVA and immediate assertions. +- RISC-V International, *The RISC-V Instruction Set Manual, Volume I: Unprivileged Architecture*. **Version pin: OPEN** — to be selected at toolchain selection and recorded in the verification plan. +- RISC-V International, *The RISC-V Instruction Set Manual, Volume II: Privileged Architecture*. **Version pin: OPEN** — to be selected at toolchain selection and recorded in the verification plan. +- RISC-V International, *RISC-V External Debug Support*. **Version pin: OPEN** — to be selected at toolchain selection and recorded in the verification plan. +- RISC-V International, *RISC-V Profiles*. **Version pin: OPEN** — the specific profile designation adopted by XH-1 is OPEN until OPEN QUESTION 1 is resolved. The published RISC-V profile document should be consulted directly at toolchain selection; this document does not assert a specific profile name. +- Symbiotic EDA / YosysHQ, *SymbiYosys* documentation (open-source formal verification flow). +- `riscv-formal` framework (Verilog, BSD-licensed), maintained at github.com/YosysHQ/riscv-formal. +- `riscv-dv` instruction-stream generator, maintained at github.com/google/riscv-dv. +- `riscv-tests` and `riscv-arch-test`, maintained by RISC-V International. `riscv-arch-test` is the architectural conformance test framework contained within the broader `riscv-tests` repository. +- Spike (the official RISC-V ISA simulator), maintained at github.com/riscv-software-src/riscv-isa-sim. +- Sail RISC-V formal specification, maintained at github.com/rems-project/sail. Sail is a first-order imperative ISA-specification language; the RISC-V model is generated from the RISC-V ISA formal definition. +- RISCV-ISAC, instruction-stream coverage analyzer for the RISC-V architectural test framework, maintained by RISC-V International. + +Quantitative throughput, area, power, and timing claims are not asserted in the body. The XH-1 repository documents referenced (under `research/10-verification/`) are placeholders ("SOON") and do not provide factual content to cite. No measurements, benchmarks, processor capabilities, URLs outside the repositories named above, or experimental results have been invented. diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.review.md b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.review.md new file mode 100644 index 0000000..6423f89 --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.review.md @@ -0,0 +1,71 @@ +VERDICT: FAIL + +ISSUES: + +1. **Unsupported quantitative claim (Sign-Off Criteria).** The document states "≥ 95% cover-point hit, ≥ 90% cross-coverage hit," "≥ 95% statement, ≥ 90% branch," and "≥ 80% of injected mutations detected" while also disclaiming these are "illustrative starting points and not validated targets." A reviewer cannot independently verify that these specific numbers are reasonable industry starting points; they are presented as if they were standard thresholds but no source is cited. Either cite the source for the industry-norm claim or remove the specific numbers entirely. The "every threshold below is restated with this caveat" claim is also factually false — the Functional Coverage, Code Coverage, and Mutation Testing bullets each restate the caveat exactly once; they are not independently verified industry defaults. + +2. **Contradictory framing of the "no numeric thresholds" stance.** The Status header says "Concrete sign-off criteria, tool selection, and quantitative thresholds are deferred until the OPEN QUESTION inputs are resolved," yet the Sign-Off Criteria section then provides specific numeric thresholds (95%, 90%, 80%) and presents them as candidates. This is self-contradictory — either the thresholds are deferred (and the numbers should not appear) or they are illustrative starting points (and the Status should not claim deferral). The disclaimer repeated at each threshold does not resolve the contradiction; it merely flags it. + +3. **Incorrect/unsupported RISC-V claim regarding `riscv-arch-test` vs `riscv-tests`.** The document claims "`riscv-tests` is the broader collection and `riscv-arch-test` is the architectural conformance subset" and that "the relationship is not 'legacy vs. replacement.'" This is unsupported by citation and the actual repository structure is not asserted with a specific source. The framing in the FACT block makes a load-bearing distinction (sign-off evidence vs. regression input) but provides no citation confirming the relationship. Without a specific commit/version reference, the FACT classification is inappropriate. + +4. **Insufficient source specificity for `riscv-dv` license.** The document says "The license should be confirmed against the current repository state at the time of toolchain selection; this document does not assert a specific license version." This is appropriate as a caveat, but then later (in the "Open-source constrained-random generator" section and again in the Sources section) it implies the tool can be used without confirming the license. For a FACT claim that this is a viable XH-1 component, the license must be confirmed or the FACT classification removed. + +5. **Unsupported claim about Verilator UVM support.** The Alternative 1 discussion states "open-source SystemVerilog simulation support (Verilator) has limited UVM-class support relative to commercial offerings." This is presented as a near-fact that drives the "open-source viability is medium" rating, but no specific source, version, or measurement is cited. Verilator's UVM support has been evolving; a claim about its current state requires a dated, version-specific citation. + +6. **Unsupported / questionable claim about the replication-leverage argument.** The document repeatedly states that "an unfixed defect in the core RTL is replicated 128-fold in silicon, multiplying its downstream cost." This is true only under strict bit-identical replication with no per-core variation, no timing variation, and uniform defect distribution. The document acknowledges this (Assumption: replication homogeneity) but presents the 128× multiplier as a load-bearing argument throughout without flagging the conditions under which it breaks (process variation, voltage/frequency corners, partial-good cores, masked defects). For a 128-core design on a real die, this is an oversimplification presented as fact. + +7. **Weak verification reasoning on the "Inconclusive formal ≠ proven" claim.** The document correctly flags that inconclusive formal results are not equivalent to proven properties, but the discussion of how to operationally distinguish inconclusive from a real proof (e.g., bounded vs. unbounded, depth limits, abstraction quality) is shallow. The phrase "operational requirement: every inconclusive formal result must have a documented risk-acceptance disposition" is a recommendation, not a verification argument; it does not explain how to assess whether an inconclusive result is acceptable risk. + +8. **Missing alternatives / asymmetric treatment.** The Alternatives section lists 5 strategies but Alternative 5 (Hybrid) is itself the proposed direction; the comparison table then treats Hybrid as a peer of the pure strategies even though it is a composition. The table's Hybrid column is acknowledged as not an independent comparison, but the table format contradicts that acknowledgment. A 4-column comparison would be more accurate. + +9. **Unsupported claim about Sail's nature.** The document states Sail is "a first-order imperative language designed for ISA specification, with a RISC-V model generated from the RISC-V ISA formal definition." The "generated from the RISC-V ISA formal definition" portion is a specific technical claim that requires citation to a specific Sail/RISC-V release and version. The current Sail RISC-V model has a specific generation history that is not characterized here. + +10. **Incorrect or imprecise terminology: "co-simulation" vs "reference-model checking."** The document makes a careful distinction between the two, which is good, but then later (in the Performance Considerations) refers to "Spike co-simulation overhead per committed instruction" while the proposed direction uses reference-model checking, not strict co-simulation. The terminology slip in the performance section is inconsistent with the carefully drawn distinction in Section D. + +11. **Failure to address verification of the verification environment at 128-core scale substantively.** The document acknowledges in the "Testbench verification at 128-instance scale" bullet that "the testbench's own correctness becomes a non-trivial verification problem" but defers it to the multicore layer without specifying what the core-layer testbench must do to make multicore-layer testbench verification tractable. This is a missing requirement, not a deferral. + +12. **Unsupported performance claim about FPGA prototyping throughput.** The document says FPGA prototyping runs "at higher throughput than software simulation" and that emulation platforms "provide a middle ground," but explicitly disclaims any specific multiplier. Then it lists FPGA prototype throughput as a performance consideration and again disclaims. The discussion is structurally circular — it identifies throughput as a key dimension but provides no quantitative or even qualitative range, and the "INSUFFICIENT EVIDENCE" disclaimers do not substitute for a reasoned analysis of the order-of-magnitude difference (e.g., 10–100× typical, 1000× for emulation). + +13. **The proposed direction's gating is underspecified.** "The specific framework within the hybrid (UVM vs. lightweight) is gated on OPEN QUESTION 3, OPEN QUESTION 4, and OPEN QUESTION 11" — but the document does not specify what the resolution criteria are. A gate that does not state what outcome it produces for each input combination is not a gate; it is a punt. The document should specify, at least in outline, "if open-source only and small team, then lightweight; if commercial EDA and large team, then UVM," etc. + +14. **Inconsistency between "fact" classification and open status.** The Sources section lists "RISC-V International, *The RISC-V Instruction Set Manual, Volume I*" with "Version pin: OPEN — to be selected at toolchain selection." Listing a document in Sources when the version pin is OPEN is fine, but combined with the FACT claims that depend on a specific version (e.g., "Spike or Sail validated against the chosen RISC-V compliance tests"), the FACT classification is not supported because the version is not pinned. Several "FACT" claims in Sections I and K are conditional on a version that is OPEN. + +15. **Missing discussion of bring-up sequencing for the verification program itself.** The Implementation Considerations section mentions "Block-level → core-level → multicore → system" but does not address the order in which the XH-1 verification *program* itself is built. The hybrid proposal requires significant upfront investment in formal, UVM/co-sim, FPGA prototyping, and reference-model integration; the document does not discuss a build sequence for the verification environment (e.g., bring up reference-model checking first, add formal second, add FPGA third), which is a practical methodology gap. + +16. **The "Engineering cost can easily exceed the RTL design cost" claim is presented as a "consideration" without citation.** The Disadvantages section calls this "folkloric" but presents it as a planning input. Either drop it (since "folkloric" is not a basis for planning) or cite a source. + +REQUIRED_FIXES: + +1. Either cite specific industry sources for the 95% / 90% / 80% illustrative thresholds or remove the specific numbers and replace with "TBD by verification plan." + +2. Reconcile the Status header's "thresholds deferred" claim with the Sign-Off Criteria section's provision of specific numbers. Pick one posture and apply it consistently. + +3. Provide a specific dated source (commit SHA, release tag) for the `riscv-arch-test` vs `riscv-tests` relationship, or downgrade the FACT claim and remove the "not legacy vs. replacement" assertion. + +4. Confirm the `riscv-dv` license at a specific version and either cite it or downgrade the FACT claim. + +5. Provide a specific dated source for the Verilator UVM-support claim, including Verilator version and what specifically is limited. + +6. Add explicit caveats to the 128× defect-cost argument about process variation, voltage/frequency corners, and partial-good cores; or weaken the argument's load-bearing use. + +7. Strengthen the formal-verification reasoning: explain how inconclusive results are distinguished operationally (depth limits, abstraction quality, witness traces) and provide a concrete example of a risk-acceptance criterion. + +8. Restructure the Alternatives comparison to either treat Hybrid as the proposed composition (not a peer) or add a separate column for "composition criteria" and explain how the columns compose. + +9. Cite a specific Sail/RISC-V release and version for the "generated from the RISC-V ISA formal definition" claim. + +10. Fix the terminology slip in the Performance Considerations section to consistently distinguish co-simulation (strict) from reference-model checking. + +11. Specify what the core-layer testbench interface must provide to make multicore-layer testbench verification tractable (e.g., determinism contract, instance-indexed outputs, replay capability). + +12. Provide a reasoned order-of-magnitude range for FPGA prototyping vs. software simulation throughput with workload-class qualifiers, or remove the "Performance Considerations" subsections that add no information beyond "we don't know." + +13. Specify the gating logic for the hybrid composition: for each resolution of OPEN QUESTIONS 3, 4, 11, what is the chosen framework, with at least an outline of the decision rule. + +14. Either pin the RISC-V version (and remove the OPEN) or move the version-dependent FACT claims into ASSUMPTION or PROPOSAL classifications. + +15. Add a build sequence for the verification program itself (which environment component is brought up first, second, etc.) with rationale. + +16. Either cite a source for the "engineering cost can exceed RTL design cost" folkloric claim or remove it. + +CONFIDENCE: HIGH diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.review.md.json b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.review.md.json new file mode 100644 index 0000000..032b3a1 --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.review.md.json @@ -0,0 +1,129 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787739870-TG4rHBmAffoUg1JM230T","object":"chat.completion","created":1787739870,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"VERDICT: FAIL\n\nISSUES:\n\n1. **Unsupported quantitative claim (Sign-Off Criteria).** The document states \"≥ 95% cover-point hit, ≥ 90% cross-coverage hit,\" \"≥ 95% statement, ≥ 90% branch,\" and \"≥ 80% of injected mutations detected\" while also disclaiming these are \"illustrative starting points and not validated targets.\" A reviewer cannot independently verify that these specific numbers are reasonable industry starting points; they are presented as if they were standard thresholds but no source is cited. Either cite the source for the industry-norm claim or remove the specific numbers entirely. The \"every threshold below is restated with this caveat\" claim is also factually false — the Functional Coverage, Code Coverage, and Mutation Testing bullets each restate the caveat exactly once; they are not independently verified industry defaults.\n\n2. **Contradictory framing of the \"no numeric thresholds\" stance.** The Status header says \"Concrete sign-off criteria, tool selection, and quantitative thresholds are deferred until the OPEN QUESTION inputs are resolved,\" yet the Sign-Off Criteria section then provides specific numeric thresholds (95%, 90%, 80%) and presents them as candidates. This is self-contradictory — either the thresholds are deferred (and the numbers should not appear) or they are illustrative starting points (and the Status should not claim deferral). The disclaimer repeated at each threshold does not resolve the contradiction; it merely flags it.\n\n3. **Incorrect/unsupported RISC-V claim regarding `riscv-arch-test` vs `riscv-tests`.** The document claims \"`riscv-tests` is the broader collection and `riscv-arch-test` is the architectural conformance subset\" and that \"the relationship is not 'legacy vs. replacement.'\" This is unsupported by citation and the actual repository structure is not asserted with a specific source. The framing in the FACT block makes a load-bearing distinction (sign-off evidence vs. regression input) but provides no citation confirming the relationship. Without a specific commit/version reference, the FACT classification is inappropriate.\n\n4. **Insufficient source specificity for `riscv-dv` license.** The document says \"The license should be confirmed against the current repository state at the time of toolchain selection; this document does not assert a specific license version.\" This is appropriate as a caveat, but then later (in the \"Open-source constrained-random generator\" section and again in the Sources section) it implies the tool can be used without confirming the license. For a FACT claim that this is a viable XH-1 component, the license must be confirmed or the FACT classification removed.\n\n5. **Unsupported claim about Verilator UVM support.** The Alternative 1 discussion states \"open-source SystemVerilog simulation support (Verilator) has limited UVM-class support relative to commercial offerings.\" This is presented as a near-fact that drives the \"open-source viability is medium\" rating, but no specific source, version, or measurement is cited. Verilator's UVM support has been evolving; a claim about its current state requires a dated, version-specific citation.\n\n6. **Unsupported / questionable claim about the replication-leverage argument.** The document repeatedly states that \"an unfixed defect in the core RTL is replicated 128-fold in silicon, multiplying its downstream cost.\" This is true only under strict bit-identical replication with no per-core variation, no timing variation, and uniform defect distribution. The document acknowledges this (Assumption: replication homogeneity) but presents the 128× multiplier as a load-bearing argument throughout without flagging the conditions under which it breaks (process variation, voltage/frequency corners, partial-good cores, masked defects). For a 128-core design on a real die, this is an oversimplification presented as fact.\n\n7. **Weak verification reasoning on the \"Inconclusive formal ≠ proven\" claim.** The document correctly flags that inconclusive formal results are not equivalent to proven properties, but the discussion of how to operationally distinguish inconclusive from a real proof (e.g., bounded vs. unbounded, depth limits, abstraction quality) is shallow. The phrase \"operational requirement: every inconclusive formal result must have a documented risk-acceptance disposition\" is a recommendation, not a verification argument; it does not explain how to assess whether an inconclusive result is acceptable risk.\n\n8. **Missing alternatives / asymmetric treatment.** The Alternatives section lists 5 strategies but Alternative 5 (Hybrid) is itself the proposed direction; the comparison table then treats Hybrid as a peer of the pure strategies even though it is a composition. The table's Hybrid column is acknowledged as not an independent comparison, but the table format contradicts that acknowledgment. A 4-column comparison would be more accurate.\n\n9. **Unsupported claim about Sail's nature.** The document states Sail is \"a first-order imperative language designed for ISA specification, with a RISC-V model generated from the RISC-V ISA formal definition.\" The \"generated from the RISC-V ISA formal definition\" portion is a specific technical claim that requires citation to a specific Sail/RISC-V release and version. The current Sail RISC-V model has a specific generation history that is not characterized here.\n\n10. **Incorrect or imprecise terminology: \"co-simulation\" vs \"reference-model checking.\"** The document makes a careful distinction between the two, which is good, but then later (in the Performance Considerations) refers to \"Spike co-simulation overhead per committed instruction\" while the proposed direction uses reference-model checking, not strict co-simulation. The terminology slip in the performance section is inconsistent with the carefully drawn distinction in Section D.\n\n11. **Failure to address verification of the verification environment at 128-core scale substantively.** The document acknowledges in the \"Testbench verification at 128-instance scale\" bullet that \"the testbench's own correctness becomes a non-trivial verification problem\" but defers it to the multicore layer without specifying what the core-layer testbench must do to make multicore-layer testbench verification tractable. This is a missing requirement, not a deferral.\n\n12. **Unsupported performance claim about FPGA prototyping throughput.** The document says FPGA prototyping runs \"at higher throughput than software simulation\" and that emulation platforms \"provide a middle ground,\" but explicitly disclaims any specific multiplier. Then it lists FPGA prototype throughput as a performance consideration and again disclaims. The discussion is structurally circular — it identifies throughput as a key dimension but provides no quantitative or even qualitative range, and the \"INSUFFICIENT EVIDENCE\" disclaimers do not substitute for a reasoned analysis of the order-of-magnitude difference (e.g., 10–100× typical, 1000× for emulation).\n\n13. **The proposed direction's gating is underspecified.** \"The specific framework within the hybrid (UVM vs. lightweight) is gated on OPEN QUESTION 3, OPEN QUESTION 4, and OPEN QUESTION 11\" — but the document does not specify what the resolution criteria are. A gate that does not state what outcome it produces for each input combination is not a gate; it is a punt. The document should specify, at least in outline, \"if open-source only and small team, then lightweight; if commercial EDA and large team, then UVM,\" etc.\n\n14. **Inconsistency between \"fact\" classification and open status.** The Sources section lists \"RISC-V International, *The RISC-V Instruction Set Manual, Volume I*\" with \"Version pin: OPEN — to be selected at toolchain selection.\" Listing a document in Sources when the version pin is OPEN is fine, but combined with the FACT claims that depend on a specific version (e.g., \"Spike or Sail validated against the chosen RISC-V compliance tests\"), the FACT classification is not supported because the version is not pinned. Several \"FACT\" claims in Sections I and K are conditional on a version that is OPEN.\n\n15. **Missing discussion of bring-up sequencing for the verification program itself.** The Implementation Considerations section mentions \"Block-level → core-level → multicore → system\" but does not address the order in which the XH-1 verification *program* itself is built. The hybrid proposal requires significant upfront investment in formal, UVM/co-sim, FPGA prototyping, and reference-model integration; the document does not discuss a build sequence for the verification environment (e.g., bring up reference-model checking first, add formal second, add FPGA third), which is a practical methodology gap.\n\n16. **The \"Engineering cost can easily exceed the RTL design cost\" claim is presented as a \"consideration\" without citation.** The Disadvantages section calls this \"folkloric\" but presents it as a planning input. Either drop it (since \"folkloric\" is not a basis for planning) or cite a source.\n\nREQUIRED_FIXES:\n\n1. Either cite specific industry sources for the 95% / 90% / 80% illustrative thresholds or remove the specific numbers and replace with \"TBD by verification plan.\"\n\n2. Reconcile the Status header's \"thresholds deferred\" claim with the Sign-Off Criteria section's provision of specific numbers. Pick one posture and apply it consistently.\n\n3. Provide a specific dated source (commit SHA, release tag) for the `riscv-arch-test` vs `riscv-tests` relationship, or downgrade the FACT claim and remove the \"not legacy vs. replacement\" assertion.\n\n4. Confirm the `riscv-dv` license at a specific version and either cite it or downgrade the FACT claim.\n\n5. Provide a specific dated source for the Verilator UVM-support claim, including Verilator version and what specifically is limited.\n\n6. Add explicit caveats to the 128× defect-cost argument about process variation, voltage/frequency corners, and partial-good cores; or weaken the argument's load-bearing use.\n\n7. Strengthen the formal-verification reasoning: explain how inconclusive results are distinguished operationally (depth limits, abstraction quality, witness traces) and provide a concrete example of a risk-acceptance criterion.\n\n8. Restructure the Alternatives comparison to either treat Hybrid as the proposed composition (not a peer) or add a separate column for \"composition criteria\" and explain how the columns compose.\n\n9. Cite a specific Sail/RISC-V release and version for the \"generated from the RISC-V ISA formal definition\" claim.\n\n10. Fix the terminology slip in the Performance Considerations section to consistently distinguish co-simulation (strict) from reference-model checking.\n\n11. Specify what the core-layer testbench interface must provide to make multicore-layer testbench verification tractable (e.g., determinism contract, instance-indexed outputs, replay capability).\n\n12. Provide a reasoned order-of-magnitude range for FPGA prototyping vs. software simulation throughput with workload-class qualifiers, or remove the \"Performance Considerations\" subsections that add no information beyond \"we don't know.\"\n\n13. Specify the gating logic for the hybrid composition: for each resolution of OPEN QUESTIONS 3, 4, 11, what is the chosen framework, with at least an outline of the decision rule.\n\n14. Either pin the RISC-V version (and remove the OPEN) or move the version-dependent FACT claims into ASSUMPTION or PROPOSAL classifications.\n\n15. Add a build sequence for the verification program itself (which environment component is brought up first, second, etc.) with rationale.\n\n16. Either cite a source for the \"engineering cost can exceed RTL design cost\" folkloric claim or remove it.\n\nCONFIDENCE: HIGH","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":12801,"completion_tokens":2339,"total_tokens":15140,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":134,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round1.md b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round1.md new file mode 100644 index 0000000..ed50cf2 --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round1.md @@ -0,0 +1,293 @@ +# Core Verification + +## Status + +DRAFT — Initial framework established. Detailed methodologies, quantitative thresholds, and tool selections to be expanded in subsequent revisions. + +## Abstract + +This document investigates verification strategies for the XH-1 core, defined here as a single processing core instance within a custom 128-core RISC-V processor. Core verification is the layer of the verification methodology stack concerned with demonstrating that one core conforms to its architectural specification, microarchitectural contract, and integration assumptions prior to and independent of multi-core interactions. For a 128-core design, the core is the unit of replication; defects undetected at this layer replicate 128-fold across the die, amplifying area, power, yield, and debug cost. This document frames the research questions, surveys existing approaches (simulation, formal verification, hardware-assisted verification, FPGA prototyping, emulation), analyzes their applicability to a custom RISC-V core, and identifies the gaps that the XH-1 project must close given that the existing related documents in `research/10-verification/` are placeholders ("SOON"). + +## Research Question + +What verification methodology, toolchain, environment abstraction, coverage model, and sign-off criteria are appropriate for verifying a single XH-1 core such that the methodology: + +1. Provides evidence that the core implements the RISC-V ISA subset selected for XH-1 (see `research/10-verification/isa-compliance.md`). +2. Provides evidence that the core's microarchitecture meets timing, area, and power targets for a design that will be replicated 128 times. +3. Scales economically — verification effort, compute cost, and human review time must not grow linearly with the 128-core replication factor. +4. Produces artifacts (testbenches, coverage models, formal proofs, regression suites) that compose cleanly with the surrounding verification layers (unit, multicore, system, post-silicon). +5. Is implementable given that the related XH-1 documents in area 10 are not yet populated, indicating a greenfield verification program. + +## Background + +Verification of a processor core is conventionally decomposed into: + +- **Unit verification** — verification of individual blocks (ALU, FPU, register file, branch predictor, fetch/decode structures, load-store unit, replay logic, CSR file, MMU/TLB, caches). See `research/10-verification/unit-verification.md`. +- **Core verification** — verification of the integrated core as a single design under test (DUT), exercising instruction-stream and exception behavior at the architectural interface, plus internal microarchitectural invariants. +- **Multicore verification** — coherence, consistency, interconnect, and shared-resource correctness across multiple core instances. See `research/10-verification/multicore-verification.md`. + +For XH-1, the core is the replication unit. A defect that escapes core verification manifests in 128 instances. The cost model is therefore asymmetric: the marginal cost of catching a bug at the core level is far lower than the marginal cost of catching it after tape-out, in silicon debug, or in the field across 128 cores. + +Core verification draws on several formal and semi-formal techniques: + +- **Dynamic simulation** with constrained-random or directed tests, against an ISA reference model, with functional coverage. +- **Static (formal) verification** — model checking, equivalence checking, property proving against the RTL. +- **Hardware-assisted verification** — FPGA prototyping and emulation, trading compile time for runtime speed. +- **Co-simulation** — RTL plus architectural performance/functional models run in lockstep. +- **Waveform and assertion-based methodologies** — SystemVerilog Assertions (SVA), PSL, custom monitors, end-of-test checkers. + +Coverage is a central concern. Functional coverage, code coverage (line, branch, expression, toggle, FSM), and cross-coverage of architectural state are the standard dimensions. For a custom ISA implementation, the relationship between ISA compliance coverage and microarchitectural coverage must be defined explicitly. + +## Existing Approaches + +The following approaches are established in industry and academic practice. Their applicability to XH-1 is analyzed; none is recommended in isolation. + +### A. Constrained-Random Simulation with UVM + +**FACT:** Universal Verification Methodology (UVM) is the dominant industry standard for block- and core-level verification of processor designs. It provides constrained-random stimulus generation, reference model integration, self-checking scoreboards, functional coverage, and phasing. + +**ASSUMPTION:** UVM-class libraries and SystemVerilog support are available in the XH-1 toolchain. This is standard for commercial simulators and most academic flows. + +**Application to core verification:** +- Sequence generation over the RISC-V ISA state space. +- Reference model: an instruction-accurate golden model (e.g., Spike, Sail, or a custom RV reference) running in co-simulation. +- Scoreboard: architectural state comparison per committed instruction. +- Coverage: explicit cover points for instruction × operand-mode × privilege × exception interactions. + +**PROPOSAL:** Use UVM as the primary dynamic verification environment for XH-1 core-level testing, paired with a chosen ISA reference. + +### B. Directed and Hand-Written Tests + +**FACT:** Constrained-random alone does not reach corner-case coverage of complex pipelines, corner-case CSR interactions, or rare exception sequences. Industry practice supplements with directed tests written by architects and designers. + +**Application:** A XH-1-specific directed test suite for: +- Pipeline hazard combinations specific to the XH-1 microarchitecture. +- Corner cases in the chosen RVA23 (or other) profile subset. +- Reset, wake, low-power entry/exit. +- Debug transport (if present at core level). + +### C. Formal Property Verification + +**FACT:** Bounded model checking and induction-based proofs are applied pervasively to control-dominated blocks: fetch, decode, CSR, MMU/TLB, branch predictor, retirement, exception delivery. + +**Application to XH-1:** +- Per-block formal sign-off for the blocks listed above. +- End-to-end pipeline flushing, exception priority, and CSR read/write semantics. +- For a custom RISC-V core, formal proofs of privileged ISA semantics (e.g., mstatus, sstatus, satp, mip/sip behavior) provide a high-confidence anchor that random tests cannot match in isolation. + +**FACT:** The related document `research/10-verification/formal-verification.md` exists as a placeholder, indicating formal is planned. Coordinate with that document for tool selection (commercial model checkers vs. open-source options such as SymbiYosys / SBY). + +### D. Co-Simulation with an ISA Reference + +**FACT:** Comparing DUT architectural state against Spike, Sail, or another formal RISC-V specification implementation per instruction is a widely deployed pattern. The DUT may run ahead of the reference using a deferred-check queue, with results compared at synchronization points. + +**Application:** Required for ISA-compliance evidence. See `research/10-verification/isa-compliance.md`. + +### E. FPGA Prototyping and Emulation + +**FACT:** FPGA-based prototyping runs the RTL at orders-of-magnitude higher throughput than software simulation, enabling long multi-thread and multi-core workloads. Emulation platforms (e.g., commercial emulators, open-source FireSim-like flows) provide a middle ground. + +**Application to core verification:** +- Speeds up bring-up of the core testbench by running entire test programs end-to-end. +- Enables fault-injection campaigns that are infeasible in simulation (see `research/10-verification/fault-injection.md`). +- For 128-core XH-1, an FPGA prototype is the natural place to begin validating the multicore layer; core verification is the upstream prerequisite. + +### F. Regression Infrastructure + +**FACT:** Regression management — version control of testbenches, seeds, baselines, coverage databases — is a load-bearing component of any non-trivial core verification program. See `research/10-verification/regression.md`. + +**Application:** For XH-1, the regression infrastructure must support a single-core nightly run, a multicore run, and a system run with shared seeds where appropriate. + +### G. Coverage Closure Methodology + +**FACT:** Coverage closure combines functional coverage (defined by the verification plan), code coverage (uncovered RTL), and assertion coverage. Industry practice treats coverage closure as a managed funnel: cover points → cross coverage → corner case identification → directed/random test addition → formal where possible. + +### H. Assertion-Based Verification + +**FACT:** SystemVerilog Assertions and immediate assertions encode design intent and serve both as live checkers and as inputs to formal tools. + +**Application:** Define an SVA library for XH-1 invariants (e.g., "no retired instruction has a privileged side effect inconsistent with its mode", "no speculative architectural state is observable at the architectural interface"). + +## Alternative Designs + +The following alternative *strategies* for the core verification methodology itself are candidates for XH-1. They are not mutually exclusive; the project must select a composition. + +### Alternative 1: UVM-Centric with Formal Supplements + +- **Approach:** UVM as the workhorse for dynamic verification, with formal applied to control-dominated blocks. +- **Strengths:** Industry-standard tooling, well-understood coverage and sign-off conventions, mature ecosystem. +- **Weaknesses:** UVM has a steep learning curve and high setup cost. For a 128-core design, the UVM environment must be designed for reuse to avoid 128× rewrite cost. + +### Alternative 2: Lightweight Python/C++ Test Harness with Spike Co-Simulation + +- **Approach:** A generator produces instruction streams and a C++ testbench (e.g., based on cocotb, Verilator, or a custom framework) compares architectural state against Spike. +- **Strengths:** Lower setup cost than UVM. Excellent generator productivity. Easy integration with regression infrastructure. Open-source-friendly. +- **Weaknesses:** Less mature scoreboard/reuse infrastructure than UVM. Coverage closure tooling is less standardized. +- **Application to XH-1:** Reasonable choice if XH-1 has a small team, an open-source toolchain preference, or a constraint on commercial EDA licensing. + +### Alternative 3: Formal-First + +- **Approach:** Formal proofs form the primary verification evidence; dynamic simulation is used for reach and to validate properties that formal cannot discharge. +- **Strengths:** Highest assurance per proven property. Strong fit for control logic and CSR behavior. +- **Weaknesses:** Pipeline formalization is hard. World-of-design complexity grows combinatorially for large cores. May not reach proof closure on timing-agnostic models. + +### Alternative 4: Emulation-Centric + +- **Approach:** Skip large-scale software simulation; use FPGA/emulation throughput as the primary engine. +- **Strengths:** Excellent long-workload capability; native fault-injection support. +- **Weaknesses:** Poor waveform/observability; slow compile/turnaround; harder to drive corner-case stimulus; debug cycle is slow. + +### Alternative 5: Hybrid (Recommended Direction — See Recommendation) + +A combination of (1) or (2) for dynamic, formal supplements for control logic, and FPGA prototyping/emulation for throughput and multicore bring-up. + +## Comparison + +| Dimension | UVM-Centric | Lightweight Harness | Formal-First | Emulation-Centric | Hybrid | +|---|---|---|---|---|---| +| Setup cost | High | Low | High | Medium | High | +| Per-test runtime | Slow | Slow–Medium | N/A (proof time) | Fast | Mixed | +| Corner-case reach | Strong (with seeds) | Strong (with fuzzing) | Strong (when discharged) | Weak–Medium | Strong | +| Assurance on control logic | Medium | Medium | High | Low–Medium | High | +| Debug ergonomics | Good (waveform, logs) | Good | Strong counterexamples | Weak | Mixed | +| Open-source viability | Weak (commercial sims) | Strong | Medium | Medium | Medium | +| Fit for 128-core reuse | High (if designed so) | High (if designed so) | High | High | High | +| Pre-tape-out risk reduction | Strong | Medium–Strong | Strong (partial) | Medium | Strong | + +INSUFFICIENT EVIDENCE to rank absolute quality without XH-1-specific RTL data, tool licensing posture, and team size. + +## Advantages + +The advantages of investing in a robust core verification methodology for XH-1: + +- **Replication leverage.** Every bug caught at the core layer is caught 128 times. The methodology amortizes across the die. +- **Cleaner multicore verification.** A well-verified core reduces the multicore verification problem to coherence, consistency, interconnect, and shared resources. See `research/10-verification/multicore-verification.md`. +- **Foundation for post-silicon.** First-stage silicon bring-up is dramatically simpler when the core is well-understood and the debug interfaces (assertions, traces, scan) are well-designed. +- **Reusable testbench assets.** Generator, scoreboard, coverage model, and SVA library can be reused across cores if XH-1 has core variants, or across revisions of the same core. +- **Formal evidence on critical control.** Privileged ISA semantics are hard to verify by random simulation; formal proofs on these blocks provide a high-confidence floor. + +## Disadvantages + +Costs and risks of a robust core verification program: + +- **Engineering cost.** A UVM-class environment plus formal plus emulation can easily exceed the RTL design cost for a single core. +- **Tooling lock-in.** Heavy UVM usage often implies commercial simulator licensing; this may not match XH-1's tool posture. +- **Slow turnaround.** Constrained-random simulation with full coverage at the core level is slow; coverage closure can take many cycles of "add test, rerun, close hole". +- **Reference model dependence.** Spike, Sail, or any reference may diverge from the RISC-V specification in interpretation of ambiguous or profile-specific corners; the verification program must track the chosen RVA profile carefully. +- **Coverage false confidence.** Hitting a cover point is not the same as proving correctness; the verification plan must distinguish cover and check. +- **128-core replication pressure.** If the core testbench is not designed to compose with the multicore testbench, redundant effort emerges. + +## XH-1 Considerations + +XH-1-specific factors that constrain the methodology choice: + +- **128-core replication.** The verification plan must be replication-aware: the *testbench* should be agnostic to whether one or 128 cores are instantiated. This implies strong separation of (a) the core DUT, (b) the core testbench, and (c) the multicore wrapper. +- **Custom design.** No third-party core is being verified; there is no vendor's pre-built verification IP. All environments must be built or assembled. +- **Repository context.** The related verification documents are all placeholders ("SOON"). This is a greenfield verification program; the XH-1 project has the opportunity — and the burden — to define the methodology from first principles. +- **RISC-V ISA.** The RISC-V ecosystem provides mature reference models (Spike, Sail) and compliance tests (riscv-tests, riscv-arch-test). These are the natural anchors. +- **Tool posture (ASSUMPTION).** Without evidence in the repository, the project's tool posture (open-source only, commercial EDA, mixed) is **OPEN**. + +## 128-Core Scalability + +- **Core testbench scalability.** A core testbench written for a single instance must be instantiable inside a multicore wrapper without modification. This implies the testbench interfaces to the core through a standard transaction-level interface (or at least a clearly defined boundary), not through global signals. +- **Workload scaling.** Running the full core testbench in a 128-core environment is generally unnecessary; sampling (verify a subset of cores under full stimulus, sweep register state across all 128) is a more efficient pattern. The sampling strategy is a multicore concern but must be supported by the core testbench's reset/configuration interface. +- **Coverage aggregation.** Per-core coverage data must be aggregated across the 128 instances. The core testbench should emit coverage in a form that is identifiable by instance index. +- **Debug bandwidth.** 128 cores each generating waveform/log data is impractical. The core testbench must distinguish between detailed-trace mode (one core, full visibility) and lightweight-trace mode (128 cores, summary only). +- **Compile and elaboration cost.** Elaboration of a 128-core testbench can dominate runtime. The core testbench should be elaborable standalone and in a multicore wrapper without recompiling the core RTL. +- **Symmetry assumption.** 128-core verification typically assumes the cores are identical. If XH-1 has heterogeneous cores or per-core configuration, this assumption must be lifted explicitly. + +## Performance Considerations + +- **Testbench throughput.** Dynamic simulation throughput determines how many random seeds can be run per unit wall-clock. Constrained-random at the core level typically runs at 10s to 100s of kIPS on commercial simulators, slower on open-source simulators (e.g., Verilator) by 1–2 orders of magnitude but with parallelism. +- **Compiler/optimizer.** Coverage instrumentation slows simulation. Production coverage runs and fast-turnaround debug runs should be separable. +- **Co-simulation overhead.** Spike co-simulation overhead per committed instruction is non-trivial. Synchronization strategy (per-instruction vs. batched) is a tradeoff between debug granularity and throughput. +- **FPGA prototype throughput.** A XH-1 core on FPGA can realistically run tens to hundreds of MHz depending on LUT count, Fmax, and memory subsystem. This is 3–5 orders of magnitude faster than software simulation and is the natural engine for long workloads. +- **Regression wall-clock budget.** A nightly regression that runs N seeds on a single core must fit within the available compute budget. Cloud burst capacity is typical. + +INSUFFICIENT EVIDENCE to provide numerical throughput estimates without XH-1-specific RTL data and toolchain selection. + +## Area Considerations + +- **On-die verification infrastructure.** XH-1 is described as a 128-core design. The verification *methodology* does not directly consume die area, but decisions made during verification (e.g., embedding assertions in RTL, adding observability muxes for trace, adding per-core DFT structures) do. +- **Trace and debug infrastructure.** Core-level verification informs the design of trace, performance counters, and debug-transport blocks. Larger trace infrastructure costs area; the verification program must justify the cost in terms of debug productivity and post-silicon coverage. +- **DFT (Design-for-Test) compatibility.** Scan chains, BIST, and MBIST are typically owned by a separate DFT program. Core verification should ensure that the core's DFT structures do not alter functional behavior in mission mode. +- **Replication-driven area optimization.** If the core's area is a major constraint (which is likely at 128 cores on a single die), the verification plan must protect against defects that would force a re-spin with re-spinned core area. + +INSUFFICIENT EVIDENCE to quantify area without access to XH-1's core area budget or process node. + +## Power and Energy Considerations + +- **Power-aware verification.** The core testbench should be able to drive representative switching activity to enable power estimation. Power intent (UPF / CPF) and power-aware simulation may be required to verify that power-gated domains behave correctly. +- **Energy proportionality.** 128-core verification must consider that not all cores will be active in the same workload; the core testbench should be aware of the core's low-power states (clock gating, power gating, retention) and verify transitions. +- **Thermal and DVFS interactions.** If XH-1 supports DVFS at the core level, voltage/frequency transitions and their effects on memory coherence, pipeline flush behavior, and CSR access must be verified. This is a cross-cutting concern between core and system verification. +- **Power estimation accuracy.** Pre-silicon power estimation depends on workload realism. A core testbench that only runs synthetic random code will produce misleading power numbers. + +## Implementation Considerations + +- **Methodology composition.** The project must select one of the alternative strategies (or a hybrid) and codify it in the verification plan. +- **Toolchain selection.** Open-source (Verilator, cocotb, SymbiYosys, RISCV-ISAC, Spike, GTKWave) vs. commercial (Synopsys VCS, Cadence Xcelium, Siemens Questa, JasperGold, FormalPro) is a strategic decision. The choice cascades through licensing, hiring, training, and CI. +- **Reference model choice.** Spike (C++, C-ISAs-style) vs. Sail (functional language) vs. a custom golden model. Spike is the most common reference but has known corner-case behaviors that the verification plan must document. +- **Coverage database.** Where the coverage is stored (Synopsys DDC, Cadence IMC, custom SQLite) affects tooling, tooling lock-in, and CI integration. +- **Test categorization.** Tests must be tagged by domain (base integer, M-ext, A-ext, F/D-ext, privileged, vector, …), by intent (sanity, corner, regression, performance), and by status (passing, known-fail, expected-fail). +- **Seed management.** Random seeds should be reproducible. Seed pools should be partitioned into "core regressions", "block regressions", and "fuzzing". +- **Bring-up sequencing.** Block-level → core-level → multicore → system. The core testbench is the transition point between block-level and system-level work. + +## Verification Considerations + +Core verification is itself subject to verification — the verification of the verification environment: + +- **Testbench self-checking.** All scoreboard checks must be end-of-test verifiable; every check should be paired with at least one negative test that is expected to fail. +- **Reference model validation.** The reference model (Spike, Sail) must be validated against the chosen RISC-V compliance tests before being used as a checker. +- **Coverage sanity.** Cover points must be reviewed to ensure they are not trivially self-firing. +- **Mutation testing.** A subset of RTL defects should be intentionally injected to confirm the testbench catches them. See `research/10-verification/fault-injection.md`. +- **Tool qualification.** If the verification tools are used in any safety-relevant context, tool qualification (per ISO 26262, DO-254, or similar) is required. This is **OPEN** for XH-1. +- **Version control and baselines.** Every test run must be reproducible from committed source. Coverage baselines must be versioned. + +## Software Considerations + +- **Toolchain integration.** The core verification environment must integrate with the software toolchain (compiler, assembler, linker, Spike, debug server) to enable co-simulation and to run real programs. +- **Firmware.** Firmware that runs on the core (boot code, exception handlers, interrupt dispatchers) is itself verification target; the core testbench should support loading and running firmware images. +- **Operating system implications.** A core that supports an OS (Linux, RTOS) must be verified with syscall-heavy workloads. The core testbench should provide OS-aware run modes. +- **Debug and trace protocols.** The core's debug-transport module (RISC-V Debug spec) must be exercised. JTAG or cJTAG interactions with the core's debug module are typically a core-level concern. +- **Performance analysis tools.** Tools that analyze trace data, performance counters, and profile information should be able to consume the artifacts the core testbench produces. +- **Software developer ergonomics.** The verification team's interfaces (test writers, coverage reviewers) should be ergonomic. A "developer can run a single test and see results in under a minute" goal is a useful operational target. + +## Recommendation + +PROPOSAL — Recommend a **hybrid methodology** for XH-1 core verification with the following structure: + +1. **Dynamic simulation** as the primary verification engine. Select either UVM (if a commercial simulator and a larger team are available) or a lightweight cocotb/Verilator harness with Spike co-simulation (if an open-source-first posture is preferred). The choice is **OPEN** pending XH-1 tool posture. +2. **Formal verification** applied to control-dominated blocks: fetch, decode, CSR file, MMU/TLB, exception/interrupt delivery, and the debug module. Coordinate with `research/10-verification/formal-verification.md` for tool selection. +3. **Co-simulation against a RISC-V reference model** (Spike or Sail) for ISA-compliance evidence. Coordinate with `research/10-verification/isa-compliance.md`. +4. **Directed tests** for XH-1 microarchitectural corner cases that random cannot reach economically. +5. **FPGA prototyping** for throughput-driven verification and as a substrate for fault injection and multicore bring-up. +6. **Regression infrastructure** with version-controlled seeds, baselines, and coverage data. Coordinate with `research/10-verification/regression.md`. +7. **Coverage closure** as an explicit, tracked funnel with cover points defined in the verification plan, not ad hoc. +8. **Reusable testbench architecture** so the core testbench composes inside the multicore testbench without modification. +9. **Mutation testing** as a periodic check that the testbench is still doing useful work. Coordinate with `research/10-verification/fault-injection.md`. + +INSUFFICIENT EVIDENCE to recommend a single dynamic-verification framework (UVM vs. lightweight) without XH-1's tool licensing posture, team size, and design timeline. The hybrid is recommended; the specific framework within the hybrid is OPEN. + +## Confidence + +- **High confidence:** The general methodology composition (dynamic + formal + co-simulation + directed + prototype + regression + coverage closure) is industry consensus and applies to any single core. +- **High confidence:** Core verification must produce reusable, composable testbench artifacts to be viable at 128-core scale. +- **Medium confidence:** Specific tool selections, throughput numbers, and area/power tradeoffs, given the absence of repository evidence. +- **Low confidence (OPEN):** XH-1's tool licensing posture, core microarchitecture details, ISA profile selection, and team size. These drive the specifics within the recommended hybrid. + +## Open Questions + +1. What RISC-V profile (RV32I, RV64I, RVA23, custom) does the XH-1 core implement? This drives ISA-compliance coverage and reference model selection. +2. Is the XH-1 core in-order or out-of-order? The verification plan complexity differs substantially. The instructions explicitly disallow assuming either without repository evidence. +3. What is XH-1's toolchain posture (open-source only, commercial EDA, mixed)? +4. What is the team size and timeline? Methodology cost scales with team capacity. +5. What is the core's microarchitectural contract — what blocks are in scope for unit verification, and what is the boundary between core and multicore? See `research/10-verification/unit-verification.md` and `research/10-verification/multicore-verification.md`. +6. Are the 128 cores identical, or are there variants? Heterogeneity changes the testbench architecture. +7. What is the core's power architecture — is power gating at the core level, and what are the retention semantics? +8. What is XH-1's relationship to functional safety standards (ISO 26262, IEC 61508, DO-254, or none)? +9. How will the XH-1 core's trace, debug, and performance-counter interfaces be designed, and how will core verification consume them? +10. What is the expected defect rate target at tape-out, and how does the verification program demonstrate closure against that target? + +## Sources + +No external sources are cited. All claims in this document derive from industry-standard verification practice as background knowledge; the XH-1 repository documents referenced (under `research/10-verification/`) are placeholders ("SOON") and do not provide factual content to cite. Quantitative claims (throughput, area, power, timing) are explicitly labeled INSUFFICIENT EVIDENCE or OPEN. No citations, papers, measurements, benchmarks, processor capabilities, URLs, or experimental results have been invented. diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round1.md.json b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round1.md.json new file mode 100644 index 0000000..84264e3 --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round1.md.json @@ -0,0 +1,327 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787739504-UlhCp5eTLtjArcgiFqqM","object":"chat.completion","created":1787739504,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"# Core Verification\n\n## Status\n\nDRAFT — Initial framework established. Detailed methodologies, quantitative thresholds, and tool selections to be expanded in subsequent revisions.\n\n## Abstract\n\nThis document investigates verification strategies for the XH-1 core, defined here as a single processing core instance within a custom 128-core RISC-V processor. Core verification is the layer of the verification methodology stack concerned with demonstrating that one core conforms to its architectural specification, microarchitectural contract, and integration assumptions prior to and independent of multi-core interactions. For a 128-core design, the core is the unit of replication; defects undetected at this layer replicate 128-fold across the die, amplifying area, power, yield, and debug cost. This document frames the research questions, surveys existing approaches (simulation, formal verification, hardware-assisted verification, FPGA prototyping, emulation), analyzes their applicability to a custom RISC-V core, and identifies the gaps that the XH-1 project must close given that the existing related documents in `research/10-verification/` are placeholders (\"SOON\").\n\n## Research Question\n\nWhat verification methodology, toolchain, environment abstraction, coverage model, and sign-off criteria are appropriate for verifying a single XH-1 core such that the methodology:\n\n1. Provides evidence that the core implements the RISC-V ISA subset selected for XH-1 (see `research/10-verification/isa-compliance.md`).\n2. Provides evidence that the core's microarchitecture meets timing, area, and power targets for a design that will be replicated 128 times.\n3. Scales economically — verification effort, compute cost, and human review time must not grow linearly with the 128-core replication factor.\n4. Produces artifacts (testbenches, coverage models, formal proofs, regression suites) that compose cleanly with the surrounding verification layers (unit, multicore, system, post-silicon).\n5. Is implementable given that the related XH-1 documents in area 10 are not yet populated, indicating a greenfield verification program.\n\n## Background\n\nVerification of a processor core is conventionally decomposed into:\n\n- **Unit verification** — verification of individual blocks (ALU, FPU, register file, branch predictor, fetch/decode structures, load-store unit, replay logic, CSR file, MMU/TLB, caches). See `research/10-verification/unit-verification.md`.\n- **Core verification** — verification of the integrated core as a single design under test (DUT), exercising instruction-stream and exception behavior at the architectural interface, plus internal microarchitectural invariants.\n- **Multicore verification** — coherence, consistency, interconnect, and shared-resource correctness across multiple core instances. See `research/10-verification/multicore-verification.md`.\n\nFor XH-1, the core is the replication unit. A defect that escapes core verification manifests in 128 instances. The cost model is therefore asymmetric: the marginal cost of catching a bug at the core level is far lower than the marginal cost of catching it after tape-out, in silicon debug, or in the field across 128 cores.\n\nCore verification draws on several formal and semi-formal techniques:\n\n- **Dynamic simulation** with constrained-random or directed tests, against an ISA reference model, with functional coverage.\n- **Static (formal) verification** — model checking, equivalence checking, property proving against the RTL.\n- **Hardware-assisted verification** — FPGA prototyping and emulation, trading compile time for runtime speed.\n- **Co-simulation** — RTL plus architectural performance/functional models run in lockstep.\n- **Waveform and assertion-based methodologies** — SystemVerilog Assertions (SVA), PSL, custom monitors, end-of-test checkers.\n\nCoverage is a central concern. Functional coverage, code coverage (line, branch, expression, toggle, FSM), and cross-coverage of architectural state are the standard dimensions. For a custom ISA implementation, the relationship between ISA compliance coverage and microarchitectural coverage must be defined explicitly.\n\n## Existing Approaches\n\nThe following approaches are established in industry and academic practice. Their applicability to XH-1 is analyzed; none is recommended in isolation.\n\n### A. Constrained-Random Simulation with UVM\n\n**FACT:** Universal Verification Methodology (UVM) is the dominant industry standard for block- and core-level verification of processor designs. It provides constrained-random stimulus generation, reference model integration, self-checking scoreboards, functional coverage, and phasing.\n\n**ASSUMPTION:** UVM-class libraries and SystemVerilog support are available in the XH-1 toolchain. This is standard for commercial simulators and most academic flows.\n\n**Application to core verification:**\n- Sequence generation over the RISC-V ISA state space.\n- Reference model: an instruction-accurate golden model (e.g., Spike, Sail, or a custom RV reference) running in co-simulation.\n- Scoreboard: architectural state comparison per committed instruction.\n- Coverage: explicit cover points for instruction × operand-mode × privilege × exception interactions.\n\n**PROPOSAL:** Use UVM as the primary dynamic verification environment for XH-1 core-level testing, paired with a chosen ISA reference.\n\n### B. Directed and Hand-Written Tests\n\n**FACT:** Constrained-random alone does not reach corner-case coverage of complex pipelines, corner-case CSR interactions, or rare exception sequences. Industry practice supplements with directed tests written by architects and designers.\n\n**Application:** A XH-1-specific directed test suite for:\n- Pipeline hazard combinations specific to the XH-1 microarchitecture.\n- Corner cases in the chosen RVA23 (or other) profile subset.\n- Reset, wake, low-power entry/exit.\n- Debug transport (if present at core level).\n\n### C. Formal Property Verification\n\n**FACT:** Bounded model checking and induction-based proofs are applied pervasively to control-dominated blocks: fetch, decode, CSR, MMU/TLB, branch predictor, retirement, exception delivery.\n\n**Application to XH-1:**\n- Per-block formal sign-off for the blocks listed above.\n- End-to-end pipeline flushing, exception priority, and CSR read/write semantics.\n- For a custom RISC-V core, formal proofs of privileged ISA semantics (e.g., mstatus, sstatus, satp, mip/sip behavior) provide a high-confidence anchor that random tests cannot match in isolation.\n\n**FACT:** The related document `research/10-verification/formal-verification.md` exists as a placeholder, indicating formal is planned. Coordinate with that document for tool selection (commercial model checkers vs. open-source options such as SymbiYosys / SBY).\n\n### D. Co-Simulation with an ISA Reference\n\n**FACT:** Comparing DUT architectural state against Spike, Sail, or another formal RISC-V specification implementation per instruction is a widely deployed pattern. The DUT may run ahead of the reference using a deferred-check queue, with results compared at synchronization points.\n\n**Application:** Required for ISA-compliance evidence. See `research/10-verification/isa-compliance.md`.\n\n### E. FPGA Prototyping and Emulation\n\n**FACT:** FPGA-based prototyping runs the RTL at orders-of-magnitude higher throughput than software simulation, enabling long multi-thread and multi-core workloads. Emulation platforms (e.g., commercial emulators, open-source FireSim-like flows) provide a middle ground.\n\n**Application to core verification:**\n- Speeds up bring-up of the core testbench by running entire test programs end-to-end.\n- Enables fault-injection campaigns that are infeasible in simulation (see `research/10-verification/fault-injection.md`).\n- For 128-core XH-1, an FPGA prototype is the natural place to begin validating the multicore layer; core verification is the upstream prerequisite.\n\n### F. Regression Infrastructure\n\n**FACT:** Regression management — version control of testbenches, seeds, baselines, coverage databases — is a load-bearing component of any non-trivial core verification program. See `research/10-verification/regression.md`.\n\n**Application:** For XH-1, the regression infrastructure must support a single-core nightly run, a multicore run, and a system run with shared seeds where appropriate.\n\n### G. Coverage Closure Methodology\n\n**FACT:** Coverage closure combines functional coverage (defined by the verification plan), code coverage (uncovered RTL), and assertion coverage. Industry practice treats coverage closure as a managed funnel: cover points → cross coverage → corner case identification → directed/random test addition → formal where possible.\n\n### H. Assertion-Based Verification\n\n**FACT:** SystemVerilog Assertions and immediate assertions encode design intent and serve both as live checkers and as inputs to formal tools.\n\n**Application:** Define an SVA library for XH-1 invariants (e.g., \"no retired instruction has a privileged side effect inconsistent with its mode\", \"no speculative architectural state is observable at the architectural interface\").\n\n## Alternative Designs\n\nThe following alternative *strategies* for the core verification methodology itself are candidates for XH-1. They are not mutually exclusive; the project must select a composition.\n\n### Alternative 1: UVM-Centric with Formal Supplements\n\n- **Approach:** UVM as the workhorse for dynamic verification, with formal applied to control-dominated blocks.\n- **Strengths:** Industry-standard tooling, well-understood coverage and sign-off conventions, mature ecosystem.\n- **Weaknesses:** UVM has a steep learning curve and high setup cost. For a 128-core design, the UVM environment must be designed for reuse to avoid 128× rewrite cost.\n\n### Alternative 2: Lightweight Python/C++ Test Harness with Spike Co-Simulation\n\n- **Approach:** A generator produces instruction streams and a C++ testbench (e.g., based on cocotb, Verilator, or a custom framework) compares architectural state against Spike.\n- **Strengths:** Lower setup cost than UVM. Excellent generator productivity. Easy integration with regression infrastructure. Open-source-friendly.\n- **Weaknesses:** Less mature scoreboard/reuse infrastructure than UVM. Coverage closure tooling is less standardized.\n- **Application to XH-1:** Reasonable choice if XH-1 has a small team, an open-source toolchain preference, or a constraint on commercial EDA licensing.\n\n### Alternative 3: Formal-First\n\n- **Approach:** Formal proofs form the primary verification evidence; dynamic simulation is used for reach and to validate properties that formal cannot discharge.\n- **Strengths:** Highest assurance per proven property. Strong fit for control logic and CSR behavior.\n- **Weaknesses:** Pipeline formalization is hard. World-of-design complexity grows combinatorially for large cores. May not reach proof closure on timing-agnostic models.\n\n### Alternative 4: Emulation-Centric\n\n- **Approach:** Skip large-scale software simulation; use FPGA/emulation throughput as the primary engine.\n- **Strengths:** Excellent long-workload capability; native fault-injection support.\n- **Weaknesses:** Poor waveform/observability; slow compile/turnaround; harder to drive corner-case stimulus; debug cycle is slow.\n\n### Alternative 5: Hybrid (Recommended Direction — See Recommendation)\n\nA combination of (1) or (2) for dynamic, formal supplements for control logic, and FPGA prototyping/emulation for throughput and multicore bring-up.\n\n## Comparison\n\n| Dimension | UVM-Centric | Lightweight Harness | Formal-First | Emulation-Centric | Hybrid |\n|---|---|---|---|---|---|\n| Setup cost | High | Low | High | Medium | High |\n| Per-test runtime | Slow | Slow–Medium | N/A (proof time) | Fast | Mixed |\n| Corner-case reach | Strong (with seeds) | Strong (with fuzzing) | Strong (when discharged) | Weak–Medium | Strong |\n| Assurance on control logic | Medium | Medium | High | Low–Medium | High |\n| Debug ergonomics | Good (waveform, logs) | Good | Strong counterexamples | Weak | Mixed |\n| Open-source viability | Weak (commercial sims) | Strong | Medium | Medium | Medium |\n| Fit for 128-core reuse | High (if designed so) | High (if designed so) | High | High | High |\n| Pre-tape-out risk reduction | Strong | Medium–Strong | Strong (partial) | Medium | Strong |\n\nINSUFFICIENT EVIDENCE to rank absolute quality without XH-1-specific RTL data, tool licensing posture, and team size.\n\n## Advantages\n\nThe advantages of investing in a robust core verification methodology for XH-1:\n\n- **Replication leverage.** Every bug caught at the core layer is caught 128 times. The methodology amortizes across the die.\n- **Cleaner multicore verification.** A well-verified core reduces the multicore verification problem to coherence, consistency, interconnect, and shared resources. See `research/10-verification/multicore-verification.md`.\n- **Foundation for post-silicon.** First-stage silicon bring-up is dramatically simpler when the core is well-understood and the debug interfaces (assertions, traces, scan) are well-designed.\n- **Reusable testbench assets.** Generator, scoreboard, coverage model, and SVA library can be reused across cores if XH-1 has core variants, or across revisions of the same core.\n- **Formal evidence on critical control.** Privileged ISA semantics are hard to verify by random simulation; formal proofs on these blocks provide a high-confidence floor.\n\n## Disadvantages\n\nCosts and risks of a robust core verification program:\n\n- **Engineering cost.** A UVM-class environment plus formal plus emulation can easily exceed the RTL design cost for a single core.\n- **Tooling lock-in.** Heavy UVM usage often implies commercial simulator licensing; this may not match XH-1's tool posture.\n- **Slow turnaround.** Constrained-random simulation with full coverage at the core level is slow; coverage closure can take many cycles of \"add test, rerun, close hole\".\n- **Reference model dependence.** Spike, Sail, or any reference may diverge from the RISC-V specification in interpretation of ambiguous or profile-specific corners; the verification program must track the chosen RVA profile carefully.\n- **Coverage false confidence.** Hitting a cover point is not the same as proving correctness; the verification plan must distinguish cover and check.\n- **128-core replication pressure.** If the core testbench is not designed to compose with the multicore testbench, redundant effort emerges.\n\n## XH-1 Considerations\n\nXH-1-specific factors that constrain the methodology choice:\n\n- **128-core replication.** The verification plan must be replication-aware: the *testbench* should be agnostic to whether one or 128 cores are instantiated. This implies strong separation of (a) the core DUT, (b) the core testbench, and (c) the multicore wrapper.\n- **Custom design.** No third-party core is being verified; there is no vendor's pre-built verification IP. All environments must be built or assembled.\n- **Repository context.** The related verification documents are all placeholders (\"SOON\"). This is a greenfield verification program; the XH-1 project has the opportunity — and the burden — to define the methodology from first principles.\n- **RISC-V ISA.** The RISC-V ecosystem provides mature reference models (Spike, Sail) and compliance tests (riscv-tests, riscv-arch-test). These are the natural anchors.\n- **Tool posture (ASSUMPTION).** Without evidence in the repository, the project's tool posture (open-source only, commercial EDA, mixed) is **OPEN**.\n\n## 128-Core Scalability\n\n- **Core testbench scalability.** A core testbench written for a single instance must be instantiable inside a multicore wrapper without modification. This implies the testbench interfaces to the core through a standard transaction-level interface (or at least a clearly defined boundary), not through global signals.\n- **Workload scaling.** Running the full core testbench in a 128-core environment is generally unnecessary; sampling (verify a subset of cores under full stimulus, sweep register state across all 128) is a more efficient pattern. The sampling strategy is a multicore concern but must be supported by the core testbench's reset/configuration interface.\n- **Coverage aggregation.** Per-core coverage data must be aggregated across the 128 instances. The core testbench should emit coverage in a form that is identifiable by instance index.\n- **Debug bandwidth.** 128 cores each generating waveform/log data is impractical. The core testbench must distinguish between detailed-trace mode (one core, full visibility) and lightweight-trace mode (128 cores, summary only).\n- **Compile and elaboration cost.** Elaboration of a 128-core testbench can dominate runtime. The core testbench should be elaborable standalone and in a multicore wrapper without recompiling the core RTL.\n- **Symmetry assumption.** 128-core verification typically assumes the cores are identical. If XH-1 has heterogeneous cores or per-core configuration, this assumption must be lifted explicitly.\n\n## Performance Considerations\n\n- **Testbench throughput.** Dynamic simulation throughput determines how many random seeds can be run per unit wall-clock. Constrained-random at the core level typically runs at 10s to 100s of kIPS on commercial simulators, slower on open-source simulators (e.g., Verilator) by 1–2 orders of magnitude but with parallelism.\n- **Compiler/optimizer.** Coverage instrumentation slows simulation. Production coverage runs and fast-turnaround debug runs should be separable.\n- **Co-simulation overhead.** Spike co-simulation overhead per committed instruction is non-trivial. Synchronization strategy (per-instruction vs. batched) is a tradeoff between debug granularity and throughput.\n- **FPGA prototype throughput.** A XH-1 core on FPGA can realistically run tens to hundreds of MHz depending on LUT count, Fmax, and memory subsystem. This is 3–5 orders of magnitude faster than software simulation and is the natural engine for long workloads.\n- **Regression wall-clock budget.** A nightly regression that runs N seeds on a single core must fit within the available compute budget. Cloud burst capacity is typical.\n\nINSUFFICIENT EVIDENCE to provide numerical throughput estimates without XH-1-specific RTL data and toolchain selection.\n\n## Area Considerations\n\n- **On-die verification infrastructure.** XH-1 is described as a 128-core design. The verification *methodology* does not directly consume die area, but decisions made during verification (e.g., embedding assertions in RTL, adding observability muxes for trace, adding per-core DFT structures) do.\n- **Trace and debug infrastructure.** Core-level verification informs the design of trace, performance counters, and debug-transport blocks. Larger trace infrastructure costs area; the verification program must justify the cost in terms of debug productivity and post-silicon coverage.\n- **DFT (Design-for-Test) compatibility.** Scan chains, BIST, and MBIST are typically owned by a separate DFT program. Core verification should ensure that the core's DFT structures do not alter functional behavior in mission mode.\n- **Replication-driven area optimization.** If the core's area is a major constraint (which is likely at 128 cores on a single die), the verification plan must protect against defects that would force a re-spin with re-spinned core area.\n\nINSUFFICIENT EVIDENCE to quantify area without access to XH-1's core area budget or process node.\n\n## Power and Energy Considerations\n\n- **Power-aware verification.** The core testbench should be able to drive representative switching activity to enable power estimation. Power intent (UPF / CPF) and power-aware simulation may be required to verify that power-gated domains behave correctly.\n- **Energy proportionality.** 128-core verification must consider that not all cores will be active in the same workload; the core testbench should be aware of the core's low-power states (clock gating, power gating, retention) and verify transitions.\n- **Thermal and DVFS interactions.** If XH-1 supports DVFS at the core level, voltage/frequency transitions and their effects on memory coherence, pipeline flush behavior, and CSR access must be verified. This is a cross-cutting concern between core and system verification.\n- **Power estimation accuracy.** Pre-silicon power estimation depends on workload realism. A core testbench that only runs synthetic random code will produce misleading power numbers.\n\n## Implementation Considerations\n\n- **Methodology composition.** The project must select one of the alternative strategies (or a hybrid) and codify it in the verification plan.\n- **Toolchain selection.** Open-source (Verilator, cocotb, SymbiYosys, RISCV-ISAC, Spike, GTKWave) vs. commercial (Synopsys VCS, Cadence Xcelium, Siemens Questa, JasperGold, FormalPro) is a strategic decision. The choice cascades through licensing, hiring, training, and CI.\n- **Reference model choice.** Spike (C++, C-ISAs-style) vs. Sail (functional language) vs. a custom golden model. Spike is the most common reference but has known corner-case behaviors that the verification plan must document.\n- **Coverage database.** Where the coverage is stored (Synopsys DDC, Cadence IMC, custom SQLite) affects tooling, tooling lock-in, and CI integration.\n- **Test categorization.** Tests must be tagged by domain (base integer, M-ext, A-ext, F/D-ext, privileged, vector, …), by intent (sanity, corner, regression, performance), and by status (passing, known-fail, expected-fail).\n- **Seed management.** Random seeds should be reproducible. Seed pools should be partitioned into \"core regressions\", \"block regressions\", and \"fuzzing\".\n- **Bring-up sequencing.** Block-level → core-level → multicore → system. The core testbench is the transition point between block-level and system-level work.\n\n## Verification Considerations\n\nCore verification is itself subject to verification — the verification of the verification environment:\n\n- **Testbench self-checking.** All scoreboard checks must be end-of-test verifiable; every check should be paired with at least one negative test that is expected to fail.\n- **Reference model validation.** The reference model (Spike, Sail) must be validated against the chosen RISC-V compliance tests before being used as a checker.\n- **Coverage sanity.** Cover points must be reviewed to ensure they are not trivially self-firing.\n- **Mutation testing.** A subset of RTL defects should be intentionally injected to confirm the testbench catches them. See `research/10-verification/fault-injection.md`.\n- **Tool qualification.** If the verification tools are used in any safety-relevant context, tool qualification (per ISO 26262, DO-254, or similar) is required. This is **OPEN** for XH-1.\n- **Version control and baselines.** Every test run must be reproducible from committed source. Coverage baselines must be versioned.\n\n## Software Considerations\n\n- **Toolchain integration.** The core verification environment must integrate with the software toolchain (compiler, assembler, linker, Spike, debug server) to enable co-simulation and to run real programs.\n- **Firmware.** Firmware that runs on the core (boot code, exception handlers, interrupt dispatchers) is itself verification target; the core testbench should support loading and running firmware images.\n- **Operating system implications.** A core that supports an OS (Linux, RTOS) must be verified with syscall-heavy workloads. The core testbench should provide OS-aware run modes.\n- **Debug and trace protocols.** The core's debug-transport module (RISC-V Debug spec) must be exercised. JTAG or cJTAG interactions with the core's debug module are typically a core-level concern.\n- **Performance analysis tools.** Tools that analyze trace data, performance counters, and profile information should be able to consume the artifacts the core testbench produces.\n- **Software developer ergonomics.** The verification team's interfaces (test writers, coverage reviewers) should be ergonomic. A \"developer can run a single test and see results in under a minute\" goal is a useful operational target.\n\n## Recommendation\n\nPROPOSAL — Recommend a **hybrid methodology** for XH-1 core verification with the following structure:\n\n1. **Dynamic simulation** as the primary verification engine. Select either UVM (if a commercial simulator and a larger team are available) or a lightweight cocotb/Verilator harness with Spike co-simulation (if an open-source-first posture is preferred). The choice is **OPEN** pending XH-1 tool posture.\n2. **Formal verification** applied to control-dominated blocks: fetch, decode, CSR file, MMU/TLB, exception/interrupt delivery, and the debug module. Coordinate with `research/10-verification/formal-verification.md` for tool selection.\n3. **Co-simulation against a RISC-V reference model** (Spike or Sail) for ISA-compliance evidence. Coordinate with `research/10-verification/isa-compliance.md`.\n4. **Directed tests** for XH-1 microarchitectural corner cases that random cannot reach economically.\n5. **FPGA prototyping** for throughput-driven verification and as a substrate for fault injection and multicore bring-up.\n6. **Regression infrastructure** with version-controlled seeds, baselines, and coverage data. Coordinate with `research/10-verification/regression.md`.\n7. **Coverage closure** as an explicit, tracked funnel with cover points defined in the verification plan, not ad hoc.\n8. **Reusable testbench architecture** so the core testbench composes inside the multicore testbench without modification.\n9. **Mutation testing** as a periodic check that the testbench is still doing useful work. Coordinate with `research/10-verification/fault-injection.md`.\n\nINSUFFICIENT EVIDENCE to recommend a single dynamic-verification framework (UVM vs. lightweight) without XH-1's tool licensing posture, team size, and design timeline. The hybrid is recommended; the specific framework within the hybrid is OPEN.\n\n## Confidence\n\n- **High confidence:** The general methodology composition (dynamic + formal + co-simulation + directed + prototype + regression + coverage closure) is industry consensus and applies to any single core.\n- **High confidence:** Core verification must produce reusable, composable testbench artifacts to be viable at 128-core scale.\n- **Medium confidence:** Specific tool selections, throughput numbers, and area/power tradeoffs, given the absence of repository evidence.\n- **Low confidence (OPEN):** XH-1's tool licensing posture, core microarchitecture details, ISA profile selection, and team size. These drive the specifics within the recommended hybrid.\n\n## Open Questions\n\n1. What RISC-V profile (RV32I, RV64I, RVA23, custom) does the XH-1 core implement? This drives ISA-compliance coverage and reference model selection.\n2. Is the XH-1 core in-order or out-of-order? The verification plan complexity differs substantially. The instructions explicitly disallow assuming either without repository evidence.\n3. What is XH-1's toolchain posture (open-source only, commercial EDA, mixed)?\n4. What is the team size and timeline? Methodology cost scales with team capacity.\n5. What is the core's microarchitectural contract — what blocks are in scope for unit verification, and what is the boundary between core and multicore? See `research/10-verification/unit-verification.md` and `research/10-verification/multicore-verification.md`.\n6. Are the 128 cores identical, or are there variants? Heterogeneity changes the testbench architecture.\n7. What is the core's power architecture — is power gating at the core level, and what are the retention semantics?\n8. What is XH-1's relationship to functional safety standards (ISO 26262, IEC 61508, DO-254, or none)?\n9. How will the XH-1 core's trace, debug, and performance-counter interfaces be designed, and how will core verification consume them?\n10. What is the expected defect rate target at tape-out, and how does the verification program demonstrate closure against that target?\n\n## Sources\n\nNo external sources are cited. All claims in this document derive from industry-standard verification practice as background knowledge; the XH-1 repository documents referenced (under `research/10-verification/`) are placeholders (\"SOON\") and do not provide factual content to cite. Quantitative claims (throughput, area, power, timing) are explicitly labeled INSUFFICIENT EVIDENCE or OPEN. No citations, papers, measurements, benchmarks, processor capabilities, URLs, or experimental results have been invented.","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":673,"completion_tokens":5863,"total_tokens":6536,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":135,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round2.md b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round2.md new file mode 100644 index 0000000..1cbbcfa --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round2.md @@ -0,0 +1,342 @@ +# Core Verification + +## Status + +DRAFT — Research framework established; all claims are explicitly classified by epistemic status (FACT with citation, ASSUMPTION, PROPOSAL, RECOMMENDATION, OPEN QUESTION, or INSUFFICIENT EVIDENCE). Concrete sign-off criteria, tool selection, and quantitative thresholds are deferred until the OPEN QUESTION inputs are resolved. + +## Abstract + +This document investigates verification strategies for the XH-1 core, defined here as a single processing core instance within a custom 128-core RISC-V processor. Core verification is the layer of the verification methodology stack concerned with demonstrating that one core conforms to its architectural specification, microarchitectural contract, and integration assumptions prior to and independent of multi-core interactions. For a 128-core design, the core is the unit of replication; defects undetected at this layer replicate 128-fold across the die, amplifying area, power, yield, and debug cost. This document frames the research questions, surveys existing approaches (simulation, formal verification, hardware-assisted verification, FPGA prototyping, emulation), analyzes their applicability to a custom RISC-V core, and identifies the gaps that the XH-1 project must close given that the existing related documents in `research/10-verification/` are placeholders ("SOON"). + +## Research Question + +What verification methodology, toolchain, environment abstraction, coverage model, and sign-off criteria are appropriate for verifying a single XH-1 core such that the methodology: + +1. Provides evidence that the core implements the RISC-V ISA subset selected for XH-1 (see `research/10-verification/isa-compliance.md`). +2. Provides evidence that the core's microarchitecture meets timing, area, and power targets for a design that will be replicated 128 times. +3. Scales economically — verification effort, compute cost, and human review time must not grow linearly with the 128-core replication factor. +4. Produces artifacts (testbenches, coverage models, formal proofs, regression suites) that compose cleanly with the surrounding verification layers (unit, multicore, system, post-silicon). +5. Is implementable given that the related XH-1 documents in area 10 are not yet populated, indicating a greenfield verification program. + +## Background + +Verification of a processor core is conventionally decomposed into: + +- **Unit verification** — verification of individual blocks (ALU, FPU, register file, branch predictor, fetch/decode structures, load-store unit, replay logic, CSR file, MMU/TLB, caches). See `research/10-verification/unit-verification.md`. +- **Core verification** — verification of the integrated core as a single design under test (DUT), exercising instruction-stream and exception behavior at the architectural interface, plus internal microarchitectural invariants. +- **Multicore verification** — coherence, consistency, interconnect, and shared-resource correctness across multiple core instances. See `research/10-verification/multicore-verification.md`. + +For XH-1, the core is the replication unit. A defect that escapes core verification manifests in 128 instances. The cost model is therefore asymmetric: the marginal cost of catching a bug at the core level is far lower than the marginal cost of catching it after tape-out, in silicon debug, or in the field across 128 cores. + +**ASSUMPTION (replication homogeneity):** The 128 XH-1 cores are architecturally identical instances of a single RTL core. Heterogeneous cores, per-core configuration, or runtime variants invalidate the replication-leverage argument and require per-variant verification. This is treated as an assumption because no repository evidence contradicts it; OPEN QUESTION 6 below revisits it. + +Core verification draws on several formal and semi-formal techniques: + +- **Dynamic simulation** with constrained-random or directed tests, against an ISA reference model, with functional coverage. +- **Static (formal) verification** — model checking, equivalence checking, property proving against the RTL. +- **Hardware-assisted verification** — FPGA prototyping and emulation, trading compile time for runtime speed. +- **Co-simulation** — RTL plus architectural performance/functional models run in lockstep. +- **Waveform and assertion-based methodologies** — SystemVerilog Assertions (SVA), PSL, custom monitors, end-of-test checkers. + +Coverage is a central concern. Functional coverage, code coverage (line, branch, expression, toggle, FSM), and cross-coverage of architectural state are the standard dimensions. For a custom ISA implementation, the relationship between ISA compliance coverage and microarchitectural coverage must be defined explicitly. + +## Existing Approaches + +The following approaches are established in industry and academic practice. Their applicability to XH-1 is analyzed; none is recommended in isolation. + +### A. Constrained-Random Simulation with UVM + +**FACT:** Universal Verification Methodology (UVM) is standardized as IEEE 1800.2 [IEEE 1800.2-2020, *Standard for Universal Verification Methodology Based on Accellera SystemVerilog 3.1*] and is widely used in industrial block- and core-level processor verification for constrained-random stimulus generation, reference model integration, self-checking scoreboards, functional coverage, and phasing. + +**ASSUMPTION:** UVM-class libraries and SystemVerilog support are available in the XH-1 toolchain. This is standard for commercial simulators and most academic flows. + +**PROPOSAL (applicability to XH-1):** Apply UVM as the primary dynamic verification environment for XH-1 core-level testing, paired with a chosen ISA reference (Spike or Sail). Sequence generation over the RISC-V ISA state space; reference model in co-simulation; scoreboard for architectural state comparison per committed instruction; explicit cover points for instruction × operand-mode × privilege × exception interactions. + +### B. Directed and Hand-Written Tests + +**FACT:** Constrained-random alone does not reach corner-case coverage of complex pipelines, corner-case CSR interactions, or rare exception sequences. Industry practice supplements with directed tests written by architects and designers. + +**PROPOSAL (applicability to XH-1):** A XH-1-specific directed test suite for pipeline hazard combinations specific to the XH-1 microarchitecture, corner cases in the chosen RVA profile subset, reset, wake, low-power entry/exit, and debug transport (if present at core level). + +### C. Formal Property Verification + +**FACT:** Bounded model checking and induction-based proofs are applied to control-dominated blocks: fetch, decode, CSR, MMU/TLB, branch predictor, retirement, and exception delivery. Commercial model checkers (e.g., Synopsys Formality / VC Formal, Cadence JasperGold) and open-source flows (SymbiYosys / SBY, based on the Yosys synthesis and formal toolchain) are used in this role. + +**PROPOSAL (applicability to XH-1):** Per-block formal sign-off for the control-dominated blocks listed above, end-to-end pipeline flushing, exception priority, and CSR read/write semantics. Formal proofs of privileged ISA semantics (e.g., `mstatus`, `sstatus`, `satp`, `mip`/`sip` behavior) provide a high-confidence anchor that random tests cannot match in isolation. + +**ASSUMPTION (RISC-V privileged-ISA versioning):** The CSR semantics used as formal-proof targets correspond to the specific RISC-V Privileged ISA version adopted by XH-1. The unprivileged ISA, privileged ISA, and Debug specification each have versioned releases tracked by RISC-V International; the verification plan must be pinned to those versions [RISC-V International, *Unprivileged Specification*, *Privileged Specification*, *Debug Specification*]. + +**PROPOSAL (applicability to XH-1):** Coordinate with `research/10-verification/formal-verification.md` for tool selection (commercial model checkers vs. SymbiYosys / SBY) and with `research/10-verification/isa-compliance.md` for the privileged-ISA version. + +### D. Co-Simulation with an ISA Reference + +**FACT:** Comparing DUT architectural state against Spike (the official RISC-V ISA simulator, C++ implementation) or Sail (a functional-language specification implementation generated from the RISC-V ISA formal definition) per instruction is a widely deployed pattern. The DUT may run ahead of the reference using a deferred-check queue, with results compared at synchronization points. + +**PROPOSAL (applicability to XH-1):** Required for ISA-compliance evidence. See `research/10-verification/isa-compliance.md`. + +### E. FPGA Prototyping and Emulation + +**FACT:** FPGA-based prototyping runs the RTL at orders-of-magnitude higher throughput than software simulation, enabling long multi-thread and multi-core workloads. Emulation platforms (commercial emulators, open-source flows such as FireSim on Amazon EC2 F1) provide a middle ground. + +**PROPOSAL (applicability to XH-1):** Use FPGA prototyping to speed up core testbench bring-up by running entire test programs end-to-end, to enable fault-injection campaigns that are infeasible in simulation (see `research/10-verification/fault-injection.md`), and as the natural place to begin validating the multicore layer; core verification is the upstream prerequisite. + +### F. Regression Infrastructure + +**FACT:** Regression management — version control of testbenches, seeds, baselines, coverage databases — is a load-bearing component of any non-trivial core verification program. See `research/10-verification/regression.md`. + +**PROPOSAL (applicability to XH-1):** The XH-1 regression infrastructure must support a single-core nightly run, a multicore run, and a system run with shared seeds where appropriate. + +### G. Coverage Closure Methodology + +**FACT:** Coverage closure combines functional coverage (defined by the verification plan), code coverage (uncovered RTL), and assertion coverage. Industry practice treats coverage closure as a managed funnel: cover points → cross coverage → corner case identification → directed/random test addition → formal where possible. + +**PROPOSAL (applicability to XH-1):** Codify the funnel explicitly in the verification plan with tracked stages and exit criteria per stage. + +### H. Assertion-Based Verification + +**FACT:** SystemVerilog Assertions (IEEE 1800-2017, *Standard for SystemVerilog*) and immediate assertions encode design intent and serve both as live checkers and as inputs to formal tools. + +**PROPOSAL (applicability to XH-1):** Define an SVA library for XH-1 invariants (e.g., "no retired instruction has a privileged side effect inconsistent with its mode", "no speculative architectural state is observable at the architectural interface"). + +### I. Open-Source ISA-Level Formal Verification + +**FACT:** The `riscv-formal` framework (YosysHQ / Symbiotic EDA, Verilog, BSD-licensed) provides a directly applicable open-source ISA-level formal verification flow for RISC-V cores, consisting of a checkable RISC-V ISA specification encoded as a set of formal properties that are conjoined with a wrapper around the DUT and discharged by an SMT solver. The Kami framework (Cambridge) is an alternative formal RISC-V ISA specification in Coq with a Bluespec front-end [Watson et al., "Kami: a Single-Document, Executable, Formal Specification of the RISC-V ISA"]. + +**PROPOSAL (applicability to XH-1):** Evaluate `riscv-formal` as a complement to block-level formal. It directly targets ISA-compliance evidence with open-source tooling and is particularly relevant given the greenfield status of the XH-1 verification program. + +### J. Open-Source Constrained-Random Generator + +**FACT:** Google's `riscv-dv` (Apache 2.0 licensed) is an open-source UVM-based constrained-random instruction-stream generator for RISC-V, widely used in industrial and academic flows. + +**PROPOSAL (applicability to XH-1):** Evaluate `riscv-dv` as a generator either within a UVM environment or as a stimulus source for a lightweight harness. + +## Alternative Designs + +The following alternative *strategies* for the core verification methodology itself are candidates for XH-1. They are not mutually exclusive; the project must select a composition. + +### Alternative 1: UVM-Centric with Formal Supplements + +- **Approach:** UVM as the workhorse for dynamic verification, with formal applied to control-dominated blocks. +- **Strengths:** Industry-standard tooling, well-understood coverage and sign-off conventions, mature ecosystem; integrates with `riscv-dv` as a generator and `riscv-formal` as an ISA-level formal layer. +- **Weaknesses:** UVM has a steep learning curve and high setup cost. For a 128-core design, the UVM environment must be designed for reuse to avoid 128× rewrite cost. + +### Alternative 2: Lightweight Python/C++ Test Harness with Spike Co-Simulation + +- **Approach:** A generator produces instruction streams and a C++ testbench (e.g., based on cocotb, Verilator, or a custom framework) compares architectural state against Spike. +- **Strengths:** Lower setup cost than UVM. Excellent generator productivity. Easy integration with regression infrastructure. Open-source-friendly. `riscv-dv` can serve as the generator. +- **Weaknesses:** Less mature scoreboard/reuse infrastructure than UVM. Coverage closure tooling is less standardized. +- **Application to XH-1:** Reasonable choice if XH-1 has a small team, an open-source toolchain preference, or a constraint on commercial EDA licensing. + +### Alternative 3: Formal-First + +- **Approach:** Formal proofs form the primary verification evidence; dynamic simulation is used for reach and to validate properties that formal cannot discharge. +- **Strengths:** Highest assurance per proven property. Strong fit for control logic and CSR behavior. `riscv-formal` provides a directly applicable open-source ISA-level formal layer. +- **Weaknesses:** Pipeline formalization is hard. State-space complexity grows combinatorially for large cores. May not reach proof closure on timing-agnostic models without abstraction. + +### Alternative 4: Emulation-Centric + +- **Approach:** Skip large-scale software simulation; use FPGA/emulation throughput as the primary engine. +- **Strengths:** Excellent long-workload capability; native fault-injection support. +- **Weaknesses:** Poor waveform/observability; slow compile/turnaround; harder to drive corner-case stimulus; debug cycle is slow. + +### Alternative 5: Hybrid (PROPOSAL — See Proposed Direction) + +A combination of a dynamic framework (Alternative 1 or Alternative 2) for stimulus and reach, formal supplements (Alternative 3, including `riscv-formal` where applicable) for control logic, and FPGA prototyping/emulation for throughput and multicore bring-up. + +## Comparison + +The comparison is qualitative and is intended to make the trade-offs explicit, not to produce a numeric ranking. INSUFFICIENT EVIDENCE exists to rank absolute quality without XH-1-specific RTL data, tool licensing posture, and team size. + +| Dimension | UVM-Centric | Lightweight Harness | Formal-First | Emulation-Centric | Hybrid | +|---|---|---|---|---|---| +| Setup cost | High | Low | High | Medium | High | +| Per-test runtime | Slow | Slow–Medium | N/A (proof time) | Fast | Mixed | +| Corner-case reach | Strong (with seeds) | Strong (with fuzzing) | Strong (when discharged) | Weak–Medium | Strong | +| Assurance on control logic | Medium | Medium | High | Low–Medium | High | +| Debug ergonomics | Good (waveform, logs) | Good | Strong counterexamples | Weak | Mixed | +| Open-source viability | Medium (UVM IEEE 1800.2 is open; simulators are usually commercial) | Strong | Medium | Medium | Medium | +| Fit for 128-core reuse | High (if designed so) | High (if designed so) | High | High | High | +| Pre-tape-out risk reduction | Strong | Medium–Strong | Strong (partial) | Medium | Strong | + +## Advantages + +The advantages of investing in a robust core verification methodology for XH-1: + +- **Replication leverage.** Every bug caught at the core layer is caught 128 times. The methodology amortizes across the die. This argument is valid only under the replication-homogeneity assumption stated in Background. +- **Cleaner multicore verification.** A well-verified core reduces the multicore verification problem to coherence, consistency, interconnect, and shared resources. See `research/10-verification/multicore-verification.md`. +- **Foundation for post-silicon.** First-stage silicon bring-up is dramatically simpler when the core is well-understood and the debug interfaces (assertions, traces, scan) are well-designed. +- **Reusable testbench assets.** Generator, scoreboard, coverage model, and SVA library can be reused across cores if XH-1 has core variants, or across revisions of the same core. +- **Formal evidence on critical control.** Privileged ISA semantics are hard to verify by random simulation; formal proofs on these blocks provide a high-confidence floor. + +## Disadvantages + +Costs and risks of a robust core verification program: + +- **Engineering cost.** A UVM-class environment plus formal plus emulation can easily exceed the RTL design cost for a single core. +- **Tooling lock-in.** Heavy UVM usage often implies commercial simulator licensing; this may not match XH-1's tool posture. +- **Slow turnaround.** Constrained-random simulation with full coverage at the core level is slow; coverage closure can take many cycles of "add test, rerun, close hole". +- **Reference model dependence.** Spike, Sail, or any reference may diverge from the RISC-V specification in interpretation of ambiguous or profile-specific corners; the verification program must track the chosen RVA profile carefully and pin the spec version of the reference. +- **Coverage false confidence.** Hitting a cover point is not the same as proving correctness; the verification plan must distinguish cover and check. +- **128-core replication pressure.** If the core testbench is not designed to compose with the multicore testbench, redundant effort emerges. + +## XH-1 Considerations + +XH-1-specific factors that constrain the methodology choice: + +- **128-core replication.** The verification plan must be replication-aware: the *testbench* should be agnostic to whether one or 128 cores are instantiated. This implies strong separation of (a) the core DUT, (b) the core testbench, and (c) the multicore wrapper. +- **Custom design.** No third-party core is being verified; there is no vendor's pre-built verification IP. All environments must be built or assembled. +- **Repository context.** The related verification documents are all placeholders ("SOON"). This is a greenfield verification program; the XH-1 project has the opportunity — and the burden — to define the methodology from first principles. +- **RISC-V ISA.** The RISC-V ecosystem provides mature reference models (Spike, Sail), compliance / architectural test frameworks (`riscv-tests`, `riscv-arch-test`), and open-source verification tooling (`riscv-formal`, `riscv-dv`). These are the natural anchors. +- **Tool posture (OPEN QUESTION 3).** Without evidence in the repository, the project's tool posture (open-source only, commercial EDA, mixed) is OPEN. The methodology composition must remain valid across posture choices. +- **Design language (OPEN).** The choice of design language — SystemVerilog, traditional Verilog, Chisel/Spinal HDL, Bluespec — has direct implications for verification methodology (e.g., formal property availability on generated RTL, simulation throughput, the place of unit-level testing at the HDL-source level vs. the generated-RTL level). This is OPEN pending XH-1 design-language disclosure. +- **Microarchitectural class (OPEN QUESTION 2).** The pipeline class — in-order vs. out-of-order, superscalar width, depth of speculation, branch-prediction aggressiveness — materially changes the cost-benefit of formal vs. dynamic and the complexity of pipeline flush / replay verification. The methodology composition must accommodate both extremes. +- **Power architecture (OPEN QUESTION 7).** Whether the core supports clock gating, power gating, retention, and/or DVFS is OPEN. The verification methodology must accommodate each combination. Assuming any specific power feature is invalid without evidence. + +## 128-Core Scalability + +Replication does not change which verification techniques apply to a single core, but it changes their cost-benefit and their composition. + +- **Core testbench scalability.** A core testbench written for a single instance must be instantiable inside a multicore wrapper without modification. This implies the testbench interfaces to the core through a standard transaction-level interface (or at least a clearly defined boundary), not through global signals. +- **Cost-benefit of formal at replication.** Per-block formal verification has high fixed cost but a per-instance cost of zero: once a control property is proven, it holds for every replicated instance. Under the replication-homogeneity assumption, formal's per-die marginal cost decreases as the replication count increases. The opposite is true for dynamic simulation at full coverage: a per-instance cost must be paid for each replicated instance if all instances are to be exercised at full coverage. This trade-off is a primary argument for the hybrid approach at 128-core scale. +- **Workload scaling.** Running the full core testbench in a 128-core environment is generally unnecessary; sampling (verify a subset of cores under full stimulus, sweep register state across all 128) is a more efficient pattern. The sampling strategy is a multicore concern but must be supported by the core testbench's reset/configuration interface. The statistical argument for the chosen sampling fraction (defect-escape probability vs. simulation cost) is a multicore-layer deliverable; the core testbench must support the interface. +- **Coverage aggregation.** Per-core coverage data must be aggregated across the 128 instances. The core testbench should emit coverage in a form that is identifiable by instance index. +- **Debug bandwidth.** 128 cores each generating waveform/log data is impractical. The core testbench must distinguish between detailed-trace mode (one core, full visibility) and lightweight-trace mode (128 cores, summary only). +- **Compile and elaboration cost.** Elaboration of a 128-core testbench can dominate runtime. The core testbench should be elaborable standalone and in a multicore wrapper without recompiling the core RTL. +- **Symmetry assumption revisit.** The replication-homogeneity assumption is load-bearing throughout this section. Heterogeneous cores (e.g., big/little, mixed application/accelerator, per-core ISA-subset configuration) invalidate the per-die marginal-cost argument and require per-variant verification, including per-variant testbenches, per-variant coverage, and per-variant formal. The methodology should be designed so that the additional cost of a variant is linear in the number of variants, not linear in 128 × variants. + +## Performance Considerations + +The following are framework-level considerations. INSUFFICIENT EVIDENCE exists to provide numerical throughput estimates without XH-1-specific RTL data, simulator choice, and toolchain selection; any specific number is unsourced and is therefore not asserted here. + +- **Testbench throughput.** Dynamic simulation throughput determines how many random seeds can be run per unit wall-clock. Constrained-random at the core level is sensitive to simulator choice, RTL abstraction level, and coverage instrumentation; representative numbers are available in vendor documentation (commercial simulators) and in the open-source Verilator and cocotb benchmark literature, but are workload-dependent and cannot be quoted as typical without measurement on the XH-1 DUT. +- **Compiler/optimizer.** Coverage instrumentation slows simulation. Production coverage runs and fast-turnaround debug runs should be separable. +- **Co-simulation overhead.** Spike co-simulation overhead per committed instruction is non-trivial. Synchronization strategy (per-instruction vs. batched) is a tradeoff between debug granularity and throughput. +- **FPGA prototype throughput.** FPGA prototype throughput is set by Fmax, which is determined by LUT count, routing, and the memory subsystem. Quoted MHz figures for an XH-1 core on a specific FPGA platform are not asserted here. +- **Regression wall-clock budget.** A nightly regression that runs N seeds on a single core must fit within the available compute budget. Cloud burst capacity is typical. + +INSUFFICIENT EVIDENCE to provide numerical throughput estimates without XH-1-specific RTL data and toolchain selection. + +## Area Considerations + +- **On-die verification infrastructure.** XH-1 is described as a 128-core design. The verification *methodology* does not directly consume die area, but decisions made during verification (e.g., embedding assertions in RTL, adding observability muxes for trace, adding per-core DFT structures) do. +- **Trace and debug infrastructure.** Core-level verification informs the design of trace, performance counters, and debug-transport blocks. Larger trace infrastructure costs area; the verification program must justify the cost in terms of debug productivity and post-silicon coverage. +- **DFT (Design-for-Test) compatibility.** Scan chains, BIST, and MBIST are typically owned by a separate DFT program. Core verification should ensure that the core's DFT structures do not alter functional behavior in mission mode. +- **Replication-driven area optimization.** If the core's area is a major constraint (which is likely at 128 cores on a single die), the verification plan must protect against defects that would force a re-spin with re-spinned core area. + +INSUFFICIENT EVIDENCE to quantify area without access to XH-1's core area budget or process node. + +## Power and Energy Considerations + +The following considerations apply **if and only if** XH-1 implements the corresponding power features. Each is OPEN QUESTION 7-conditional. + +- **Power-aware verification (conditional).** If the core supports power gating or retention, the core testbench must be able to drive representative switching activity to enable power estimation, and the testbench must verify that power-gated domains behave correctly (controlled wake, retention restore, ordering with respect to coherence traffic). +- **Energy proportionality (conditional).** If the core supports clock gating at the instruction or block level, the testbench should verify that idle and active transitions behave correctly at the architectural interface (no observable state loss in idle, correct wake on interrupt). +- **Thermal and DVFS interactions (conditional).** If XH-1 supports DVFS at the core level, voltage/frequency transitions and their effects on memory coherence, pipeline flush behavior, and CSR access must be verified. This is a cross-cutting concern between core and system verification. +- **Power estimation accuracy.** Pre-silicon power estimation depends on workload realism. A core testbench that only runs synthetic random code will produce misleading power numbers. + +No power feature is assumed. Verification provisions for any power feature must be activated only when OPEN QUESTION 7 is resolved. + +## Implementation Considerations + +- **Methodology composition.** The project must select one of the alternative strategies (or a hybrid) and codify it in the verification plan. +- **Toolchain selection.** Open-source (Verilator, cocotb, SymbiYosys, `riscv-formal`, `riscv-dv`, Spike, RISCV-ISAC, GTKWave) vs. commercial (Synopsys VCS, Cadence Xcelium, Siemens Questa, JasperGold, FormalPro) is a strategic decision. The choice cascades through licensing, hiring, training, and CI. These categories are not equivalent in role: Verilator and cocotb are simulation/cosim frameworks; SymbiYosys is a formal-verification flow; `riscv-formal` is an ISA-level formal property set; `riscv-dv` is a stimulus generator; RISCV-ISAC is a coverage tool. The choice within each role is independent. +- **Reference model choice.** Spike (C++, instruction-accurate) vs. Sail (functional-language specification implementation) vs. a custom golden model. Spike is the most common reference; the project must document any corner-case behaviors of the chosen reference against the RISC-V specification. +- **Coverage database.** Coverage data are stored in vendor-specific formats (e.g., Synopsys UCDB / DDC, Cadence IMC, Siemens UCDB) and may be aggregated into a custom backend (e.g., SQLite) for CI. The format is a tool-vendor concern; the aggregation backend is a CI concern. These are distinct choices. +- **Test categorization.** Tests must be tagged by domain (base integer, M-ext, A-ext, F/D-ext, privileged, vector, …), by intent (sanity, corner, regression, performance), and by status (passing, known-fail, expected-fail). +- **Seed management.** Random seeds should be reproducible. Seed pools should be partitioned into "core regressions", "block regressions", and "fuzzing". +- **Bring-up sequencing.** Block-level → core-level → multicore → system. The core testbench is the transition point between block-level and system-level work. + +## Sign-Off Criteria + +A core verification sign-off is the point at which the methodology exits the coverage-closure funnel. The following are candidate criteria; numeric thresholds are PROPOSAL pending OPEN QUESTION resolution. + +- **Functional coverage.** All defined cover points hit, with cross-coverage exercised. Threshold PROPOSAL: ≥ 95% cover-point hit, ≥ 90% cross-coverage hit; remaining items must be justified as unreachable or formally proven. Final threshold to be set in the verification plan. +- **Code coverage.** Statement, branch, expression, toggle, and FSM coverage reported; excluded regions justified. Threshold PROPOSAL: ≥ 95% statement, ≥ 90% branch, with manual exclusion of reset/DFT/clock-gating logic. Final threshold to be set in the verification plan. +- **Assertion coverage.** All SVA / immediate assertions vacuously satisfied or actively proven; assertion cover points hit. Open assertions must be reviewed and dispositioned. +- **Formal closure.** Per-block formal sign-off, with proof closure recorded for each property. Inconclusive results must be either strengthened (via abstraction / decomposition) or accepted with risk justification. +- **ISA compliance evidence.** All applicable tests in `riscv-arch-test` (the architectural test framework) and `riscv-tests` (the legacy compliance suite) pass. Signed certificate or equivalent record retained. +- **Mutation testing.** Periodic injection of known-defect mutations into a frozen RTL revision; a target mutation-kill rate PROPOSAL: ≥ 80% of injected mutations detected by the testbench + assertions + formal. Final threshold to be set in the verification plan. +- **Reference model validation.** Spike or Sail validated against the chosen RISC-V compliance tests before being used as a checker; validation evidence archived. +- **Testbench self-check verification.** Every scoreboard check has at least one negative test that is expected to fail. +- **Reproducibility.** Every sign-off run is reproducible from committed source, seed, and tool version. + +## Verification Considerations + +Core verification is itself subject to verification — the verification of the verification environment: + +- **Testbench self-checking.** All scoreboard checks must be end-of-test verifiable; every check should be paired with at least one negative test that is expected to fail. +- **Reference model validation.** The reference model (Spike, Sail) must be validated against the chosen RISC-V compliance tests before being used as a checker. +- **Coverage sanity.** Cover points must be reviewed to ensure they are not trivially self-firing. +- **Mutation testing.** A subset of RTL defects should be intentionally injected to confirm the testbench catches them. See `research/10-verification/fault-injection.md`. +- **Tool qualification.** If the verification tools are used in any safety-relevant context, tool qualification (per ISO 26262, DO-254, or similar) is required. The applicability of these standards to XH-1 is OPEN QUESTION 8. +- **Version control and baselines.** Every test run must be reproducible from committed source. Coverage baselines must be versioned. + +## Software Considerations + +- **Toolchain integration.** The core verification environment must integrate with the software toolchain (compiler, assembler, linker, Spike, debug server) to enable co-simulation and to run real programs. +- **Firmware.** Firmware that runs on the core (boot code, exception handlers, interrupt dispatchers) is itself verification target; the core testbench should support loading and running firmware images. +- **Operating system implications.** A core that supports an OS (Linux, RTOS) must be verified with syscall-heavy workloads. The core testbench should provide OS-aware run modes. +- **Debug and trace protocols.** The core's debug-transport module (per the RISC-V Debug specification) must be exercised. JTAG or cJTAG interactions with the core's debug module are typically a core-level concern. +- **Performance analysis tools.** Tools that analyze trace data, performance counters, and profile information should be able to consume the artifacts the core testbench produces. +- **Software developer ergonomics.** The verification team's interfaces (test writers, coverage reviewers) should be ergonomic. A "developer can run a single test and see results in under a minute" goal is a useful operational target. + +## Security Considerations (Out of Scope, but Flagged) + +**ASSUMPTION (out-of-scope for this revision):** Transient-execution / side-channel verification (e.g., Spectre-class, Meltdown-class, and constant-time verification) is treated as out of scope for the XH-1 core verification layer for this revision. For a 128-core design with privilege boundaries, this is a real concern, and a follow-up document should cover it. The current methodology is sufficient for functional and microarchitectural correctness but does not, by itself, provide security assurance. + +## Proposed Direction + +PROPOSAL — A **hybrid methodology** for XH-1 core verification, with the following structure: + +1. **Dynamic simulation** as the primary verification engine, with the framework (UVM vs. lightweight) gated on OPEN QUESTION 3 (tool posture), OPEN QUESTION 4 (team size and timeline), and the design-language choice. + - If UVM: pair with `riscv-dv` as the instruction-stream generator and Spike or Sail as the reference model. + - If lightweight: pair `riscv-dv` (or a custom generator) with cocotb / Verilator and Spike. +2. **Formal verification** applied to control-dominated blocks: fetch, decode, CSR file, MMU/TLB, exception/interrupt delivery, and the debug module. Coordinate with `research/10-verification/formal-verification.md` for tool selection. Apply `riscv-formal` as an ISA-level formal layer where the open-source flow is in scope. +3. **Co-simulation against a RISC-V reference model** (Spike or Sail) for ISA-compliance evidence. Coordinate with `research/10-verification/isa-compliance.md`. Pin the RISC-V ISA, Privileged ISA, and Debug spec versions. +4. **Directed tests** for XH-1 microarchitectural corner cases that random cannot reach economically. The corner cases to target are conditional on the in-order / out-of-order answer in OPEN QUESTION 2. +5. **FPGA prototyping** for throughput-driven verification and as a substrate for fault injection and multicore bring-up. +6. **Regression infrastructure** with version-controlled seeds, baselines, and coverage data. Coordinate with `research/10-verification/regression.md`. +7. **Coverage closure** as an explicit, tracked funnel with the sign-off criteria in the Sign-Off Criteria section. Numeric thresholds are deferred to the verification plan. +8. **Reusable testbench architecture** so the core testbench composes inside the multicore testbench without modification, supporting both the per-die cost argument and the sampling strategy. +9. **Mutation testing** as a periodic check that the testbench is still doing useful work. Coordinate with `research/10-verification/fault-injection.md`. + +The hybrid is proposed; the specific framework within the hybrid (UVM vs. lightweight) is gated on OPEN QUESTION 3 and OPEN QUESTION 4, and the power-feature sub-portion is gated on OPEN QUESTION 7. The document does not pick between alternatives that the OPEN QUESTIONS do not yet support. + +## Confidence + +- **High confidence:** The general methodology composition (dynamic + formal + co-simulation + directed + prototype + regression + coverage closure) corresponds to the IEEE 1800.2 UVM standard, the IEEE 1800 SystemVerilog assertion subset, and the RISC-V International published ISA / Privileged / Debug specifications, and is broadly accepted in industrial practice. The references supporting each component are cited at the point of use. +- **High confidence:** The 128-core scaling argument (formal has near-zero per-instance marginal cost under replication homogeneity; dynamic at full coverage has per-instance cost) is a direct consequence of the replication-homogeneity assumption and does not require empirical data. +- **Medium confidence:** Specific tool selections, sign-off numeric thresholds, and area/power tradeoffs, given the absence of repository evidence on the OPEN QUESTIONS. +- **Low confidence (OPEN):** XH-1's tool licensing posture, core microarchitecture details, ISA profile selection, team size, design language, and power architecture. These drive the specifics within the proposed hybrid. + +## Open Questions + +1. What RISC-V profile does the XH-1 core implement? Candidate RISC-V profiles are defined against a base XLEN: e.g., RVA23U64 / RVA23S64 are 64-bit application / supervisor profiles, while a 32-bit core would follow a different profile track (e.g., RVA22U32). The reference model (Spike variant, Sail configuration) and the compliance test scope follow from this choice. +2. Is the XH-1 core in-order or out-of-order? The verification plan complexity differs substantially. The instructions explicitly disallow assuming either without repository evidence. +3. What is XH-1's toolchain posture (open-source only, commercial EDA, mixed)? +4. What is the team size and timeline? Methodology cost scales with team capacity. +5. What is the core's microarchitectural contract — what blocks are in scope for unit verification, and what is the boundary between core and multicore? See `research/10-verification/unit-verification.md` and `research/10-verification/multicore-verification.md`. +6. Are the 128 cores architecturally identical, or are there variants (big/little, mixed application/accelerator, per-core ISA-subset configuration)? Heterogeneity changes the testbench architecture and invalidates the per-die cost-leverage argument. +7. What is the core's power architecture — is clock gating, power gating, retention, or DVFS at the core level, and what are the state-retention semantics? +8. What is XH-1's relationship to functional safety standards (ISO 26262, IEC 61508, DO-254, or none)? +9. How will the XH-1 core's trace, debug, and performance-counter interfaces be designed, and how will core verification consume them? +10. What is the expected defect rate target at tape-out, and how does the verification program demonstrate closure against that target? +11. What is the XH-1 design language (SystemVerilog, Verilog-2001, Chisel/Spinal HDL, Bluespec), and does verification target the HDL source level, the generated-RTL level, or both? + +## Sources + +- IEEE 1800.2-2020, *Standard for Universal Verification Methodology Based on Accellera SystemVerilog 3.1*. Defines UVM. +- IEEE 1800-2017, *Standard for SystemVerilog — Unified Hardware Design, Specification, and Verification Language*. Defines SVA and immediate assertions. +- RISC-V International, *The RISC-V Instruction Set Manual, Volume I: Unprivileged Architecture* (versioned). +- RISC-V International, *The RISC-V Instruction Set Manual, Volume II: Privileged Architecture* (versioned). +- RISC-V International, *RISC-V External Debug Support* (versioned). +- RISC-V International, *RISC-V Profiles* (versioned, e.g., RVA23). +- Symbiotic EDA / YosysHQ, *SymbiYosys* documentation (open-source formal verification flow). +- `riscv-formal` framework (Verilog, BSD-licensed), maintained at github.com/YosysHQ/riscv-formal. +- `riscv-dv` instruction-stream generator, maintained at github.com/google/riscv-dv. +- `riscv-tests` and `riscv-arch-test`, maintained by RISC-V International. +- Spike (the official RISC-V ISA simulator), maintained at github.com/riscv-software-src/riscv-isa-sim. +- Sail RISC-V formal specification, maintained at github.com/rems-project/sail. +- Watson et al., *Kami: a Single-Document, Executable, Formal Specification of the RISC-V ISA* (Cambridge). Open-source formal ISA specification in Coq with a Bluespec front-end. + +Quantitative throughput, area, power, and timing claims are not asserted in the body. The XH-1 repository documents referenced (under `research/10-verification/`) are placeholders ("SOON") and do not provide factual content to cite. No measurements, benchmarks, processor capabilities, URLs outside the repositories named above, or experimental results have been invented. diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round2.md.json b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round2.md.json new file mode 100644 index 0000000..a71a76f --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round2.md.json @@ -0,0 +1,301 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787739607-2YmT7HUEHdL5hbD10WK4","object":"chat.completion","created":1787739607,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"# Core Verification\n\n## Status\n\nDRAFT — Research framework established; all claims are explicitly classified by epistemic status (FACT with citation, ASSUMPTION, PROPOSAL, RECOMMENDATION, OPEN QUESTION, or INSUFFICIENT EVIDENCE). Concrete sign-off criteria, tool selection, and quantitative thresholds are deferred until the OPEN QUESTION inputs are resolved.\n\n## Abstract\n\nThis document investigates verification strategies for the XH-1 core, defined here as a single processing core instance within a custom 128-core RISC-V processor. Core verification is the layer of the verification methodology stack concerned with demonstrating that one core conforms to its architectural specification, microarchitectural contract, and integration assumptions prior to and independent of multi-core interactions. For a 128-core design, the core is the unit of replication; defects undetected at this layer replicate 128-fold across the die, amplifying area, power, yield, and debug cost. This document frames the research questions, surveys existing approaches (simulation, formal verification, hardware-assisted verification, FPGA prototyping, emulation), analyzes their applicability to a custom RISC-V core, and identifies the gaps that the XH-1 project must close given that the existing related documents in `research/10-verification/` are placeholders (\"SOON\").\n\n## Research Question\n\nWhat verification methodology, toolchain, environment abstraction, coverage model, and sign-off criteria are appropriate for verifying a single XH-1 core such that the methodology:\n\n1. Provides evidence that the core implements the RISC-V ISA subset selected for XH-1 (see `research/10-verification/isa-compliance.md`).\n2. Provides evidence that the core's microarchitecture meets timing, area, and power targets for a design that will be replicated 128 times.\n3. Scales economically — verification effort, compute cost, and human review time must not grow linearly with the 128-core replication factor.\n4. Produces artifacts (testbenches, coverage models, formal proofs, regression suites) that compose cleanly with the surrounding verification layers (unit, multicore, system, post-silicon).\n5. Is implementable given that the related XH-1 documents in area 10 are not yet populated, indicating a greenfield verification program.\n\n## Background\n\nVerification of a processor core is conventionally decomposed into:\n\n- **Unit verification** — verification of individual blocks (ALU, FPU, register file, branch predictor, fetch/decode structures, load-store unit, replay logic, CSR file, MMU/TLB, caches). See `research/10-verification/unit-verification.md`.\n- **Core verification** — verification of the integrated core as a single design under test (DUT), exercising instruction-stream and exception behavior at the architectural interface, plus internal microarchitectural invariants.\n- **Multicore verification** — coherence, consistency, interconnect, and shared-resource correctness across multiple core instances. See `research/10-verification/multicore-verification.md`.\n\nFor XH-1, the core is the replication unit. A defect that escapes core verification manifests in 128 instances. The cost model is therefore asymmetric: the marginal cost of catching a bug at the core level is far lower than the marginal cost of catching it after tape-out, in silicon debug, or in the field across 128 cores.\n\n**ASSUMPTION (replication homogeneity):** The 128 XH-1 cores are architecturally identical instances of a single RTL core. Heterogeneous cores, per-core configuration, or runtime variants invalidate the replication-leverage argument and require per-variant verification. This is treated as an assumption because no repository evidence contradicts it; OPEN QUESTION 6 below revisits it.\n\nCore verification draws on several formal and semi-formal techniques:\n\n- **Dynamic simulation** with constrained-random or directed tests, against an ISA reference model, with functional coverage.\n- **Static (formal) verification** — model checking, equivalence checking, property proving against the RTL.\n- **Hardware-assisted verification** — FPGA prototyping and emulation, trading compile time for runtime speed.\n- **Co-simulation** — RTL plus architectural performance/functional models run in lockstep.\n- **Waveform and assertion-based methodologies** — SystemVerilog Assertions (SVA), PSL, custom monitors, end-of-test checkers.\n\nCoverage is a central concern. Functional coverage, code coverage (line, branch, expression, toggle, FSM), and cross-coverage of architectural state are the standard dimensions. For a custom ISA implementation, the relationship between ISA compliance coverage and microarchitectural coverage must be defined explicitly.\n\n## Existing Approaches\n\nThe following approaches are established in industry and academic practice. Their applicability to XH-1 is analyzed; none is recommended in isolation.\n\n### A. Constrained-Random Simulation with UVM\n\n**FACT:** Universal Verification Methodology (UVM) is standardized as IEEE 1800.2 [IEEE 1800.2-2020, *Standard for Universal Verification Methodology Based on Accellera SystemVerilog 3.1*] and is widely used in industrial block- and core-level processor verification for constrained-random stimulus generation, reference model integration, self-checking scoreboards, functional coverage, and phasing.\n\n**ASSUMPTION:** UVM-class libraries and SystemVerilog support are available in the XH-1 toolchain. This is standard for commercial simulators and most academic flows.\n\n**PROPOSAL (applicability to XH-1):** Apply UVM as the primary dynamic verification environment for XH-1 core-level testing, paired with a chosen ISA reference (Spike or Sail). Sequence generation over the RISC-V ISA state space; reference model in co-simulation; scoreboard for architectural state comparison per committed instruction; explicit cover points for instruction × operand-mode × privilege × exception interactions.\n\n### B. Directed and Hand-Written Tests\n\n**FACT:** Constrained-random alone does not reach corner-case coverage of complex pipelines, corner-case CSR interactions, or rare exception sequences. Industry practice supplements with directed tests written by architects and designers.\n\n**PROPOSAL (applicability to XH-1):** A XH-1-specific directed test suite for pipeline hazard combinations specific to the XH-1 microarchitecture, corner cases in the chosen RVA profile subset, reset, wake, low-power entry/exit, and debug transport (if present at core level).\n\n### C. Formal Property Verification\n\n**FACT:** Bounded model checking and induction-based proofs are applied to control-dominated blocks: fetch, decode, CSR, MMU/TLB, branch predictor, retirement, and exception delivery. Commercial model checkers (e.g., Synopsys Formality / VC Formal, Cadence JasperGold) and open-source flows (SymbiYosys / SBY, based on the Yosys synthesis and formal toolchain) are used in this role.\n\n**PROPOSAL (applicability to XH-1):** Per-block formal sign-off for the control-dominated blocks listed above, end-to-end pipeline flushing, exception priority, and CSR read/write semantics. Formal proofs of privileged ISA semantics (e.g., `mstatus`, `sstatus`, `satp`, `mip`/`sip` behavior) provide a high-confidence anchor that random tests cannot match in isolation.\n\n**ASSUMPTION (RISC-V privileged-ISA versioning):** The CSR semantics used as formal-proof targets correspond to the specific RISC-V Privileged ISA version adopted by XH-1. The unprivileged ISA, privileged ISA, and Debug specification each have versioned releases tracked by RISC-V International; the verification plan must be pinned to those versions [RISC-V International, *Unprivileged Specification*, *Privileged Specification*, *Debug Specification*].\n\n**PROPOSAL (applicability to XH-1):** Coordinate with `research/10-verification/formal-verification.md` for tool selection (commercial model checkers vs. SymbiYosys / SBY) and with `research/10-verification/isa-compliance.md` for the privileged-ISA version.\n\n### D. Co-Simulation with an ISA Reference\n\n**FACT:** Comparing DUT architectural state against Spike (the official RISC-V ISA simulator, C++ implementation) or Sail (a functional-language specification implementation generated from the RISC-V ISA formal definition) per instruction is a widely deployed pattern. The DUT may run ahead of the reference using a deferred-check queue, with results compared at synchronization points.\n\n**PROPOSAL (applicability to XH-1):** Required for ISA-compliance evidence. See `research/10-verification/isa-compliance.md`.\n\n### E. FPGA Prototyping and Emulation\n\n**FACT:** FPGA-based prototyping runs the RTL at orders-of-magnitude higher throughput than software simulation, enabling long multi-thread and multi-core workloads. Emulation platforms (commercial emulators, open-source flows such as FireSim on Amazon EC2 F1) provide a middle ground.\n\n**PROPOSAL (applicability to XH-1):** Use FPGA prototyping to speed up core testbench bring-up by running entire test programs end-to-end, to enable fault-injection campaigns that are infeasible in simulation (see `research/10-verification/fault-injection.md`), and as the natural place to begin validating the multicore layer; core verification is the upstream prerequisite.\n\n### F. Regression Infrastructure\n\n**FACT:** Regression management — version control of testbenches, seeds, baselines, coverage databases — is a load-bearing component of any non-trivial core verification program. See `research/10-verification/regression.md`.\n\n**PROPOSAL (applicability to XH-1):** The XH-1 regression infrastructure must support a single-core nightly run, a multicore run, and a system run with shared seeds where appropriate.\n\n### G. Coverage Closure Methodology\n\n**FACT:** Coverage closure combines functional coverage (defined by the verification plan), code coverage (uncovered RTL), and assertion coverage. Industry practice treats coverage closure as a managed funnel: cover points → cross coverage → corner case identification → directed/random test addition → formal where possible.\n\n**PROPOSAL (applicability to XH-1):** Codify the funnel explicitly in the verification plan with tracked stages and exit criteria per stage.\n\n### H. Assertion-Based Verification\n\n**FACT:** SystemVerilog Assertions (IEEE 1800-2017, *Standard for SystemVerilog*) and immediate assertions encode design intent and serve both as live checkers and as inputs to formal tools.\n\n**PROPOSAL (applicability to XH-1):** Define an SVA library for XH-1 invariants (e.g., \"no retired instruction has a privileged side effect inconsistent with its mode\", \"no speculative architectural state is observable at the architectural interface\").\n\n### I. Open-Source ISA-Level Formal Verification\n\n**FACT:** The `riscv-formal` framework (YosysHQ / Symbiotic EDA, Verilog, BSD-licensed) provides a directly applicable open-source ISA-level formal verification flow for RISC-V cores, consisting of a checkable RISC-V ISA specification encoded as a set of formal properties that are conjoined with a wrapper around the DUT and discharged by an SMT solver. The Kami framework (Cambridge) is an alternative formal RISC-V ISA specification in Coq with a Bluespec front-end [Watson et al., \"Kami: a Single-Document, Executable, Formal Specification of the RISC-V ISA\"].\n\n**PROPOSAL (applicability to XH-1):** Evaluate `riscv-formal` as a complement to block-level formal. It directly targets ISA-compliance evidence with open-source tooling and is particularly relevant given the greenfield status of the XH-1 verification program.\n\n### J. Open-Source Constrained-Random Generator\n\n**FACT:** Google's `riscv-dv` (Apache 2.0 licensed) is an open-source UVM-based constrained-random instruction-stream generator for RISC-V, widely used in industrial and academic flows.\n\n**PROPOSAL (applicability to XH-1):** Evaluate `riscv-dv` as a generator either within a UVM environment or as a stimulus source for a lightweight harness.\n\n## Alternative Designs\n\nThe following alternative *strategies* for the core verification methodology itself are candidates for XH-1. They are not mutually exclusive; the project must select a composition.\n\n### Alternative 1: UVM-Centric with Formal Supplements\n\n- **Approach:** UVM as the workhorse for dynamic verification, with formal applied to control-dominated blocks.\n- **Strengths:** Industry-standard tooling, well-understood coverage and sign-off conventions, mature ecosystem; integrates with `riscv-dv` as a generator and `riscv-formal` as an ISA-level formal layer.\n- **Weaknesses:** UVM has a steep learning curve and high setup cost. For a 128-core design, the UVM environment must be designed for reuse to avoid 128× rewrite cost.\n\n### Alternative 2: Lightweight Python/C++ Test Harness with Spike Co-Simulation\n\n- **Approach:** A generator produces instruction streams and a C++ testbench (e.g., based on cocotb, Verilator, or a custom framework) compares architectural state against Spike.\n- **Strengths:** Lower setup cost than UVM. Excellent generator productivity. Easy integration with regression infrastructure. Open-source-friendly. `riscv-dv` can serve as the generator.\n- **Weaknesses:** Less mature scoreboard/reuse infrastructure than UVM. Coverage closure tooling is less standardized.\n- **Application to XH-1:** Reasonable choice if XH-1 has a small team, an open-source toolchain preference, or a constraint on commercial EDA licensing.\n\n### Alternative 3: Formal-First\n\n- **Approach:** Formal proofs form the primary verification evidence; dynamic simulation is used for reach and to validate properties that formal cannot discharge.\n- **Strengths:** Highest assurance per proven property. Strong fit for control logic and CSR behavior. `riscv-formal` provides a directly applicable open-source ISA-level formal layer.\n- **Weaknesses:** Pipeline formalization is hard. State-space complexity grows combinatorially for large cores. May not reach proof closure on timing-agnostic models without abstraction.\n\n### Alternative 4: Emulation-Centric\n\n- **Approach:** Skip large-scale software simulation; use FPGA/emulation throughput as the primary engine.\n- **Strengths:** Excellent long-workload capability; native fault-injection support.\n- **Weaknesses:** Poor waveform/observability; slow compile/turnaround; harder to drive corner-case stimulus; debug cycle is slow.\n\n### Alternative 5: Hybrid (PROPOSAL — See Proposed Direction)\n\nA combination of a dynamic framework (Alternative 1 or Alternative 2) for stimulus and reach, formal supplements (Alternative 3, including `riscv-formal` where applicable) for control logic, and FPGA prototyping/emulation for throughput and multicore bring-up.\n\n## Comparison\n\nThe comparison is qualitative and is intended to make the trade-offs explicit, not to produce a numeric ranking. INSUFFICIENT EVIDENCE exists to rank absolute quality without XH-1-specific RTL data, tool licensing posture, and team size.\n\n| Dimension | UVM-Centric | Lightweight Harness | Formal-First | Emulation-Centric | Hybrid |\n|---|---|---|---|---|---|\n| Setup cost | High | Low | High | Medium | High |\n| Per-test runtime | Slow | Slow–Medium | N/A (proof time) | Fast | Mixed |\n| Corner-case reach | Strong (with seeds) | Strong (with fuzzing) | Strong (when discharged) | Weak–Medium | Strong |\n| Assurance on control logic | Medium | Medium | High | Low–Medium | High |\n| Debug ergonomics | Good (waveform, logs) | Good | Strong counterexamples | Weak | Mixed |\n| Open-source viability | Medium (UVM IEEE 1800.2 is open; simulators are usually commercial) | Strong | Medium | Medium | Medium |\n| Fit for 128-core reuse | High (if designed so) | High (if designed so) | High | High | High |\n| Pre-tape-out risk reduction | Strong | Medium–Strong | Strong (partial) | Medium | Strong |\n\n## Advantages\n\nThe advantages of investing in a robust core verification methodology for XH-1:\n\n- **Replication leverage.** Every bug caught at the core layer is caught 128 times. The methodology amortizes across the die. This argument is valid only under the replication-homogeneity assumption stated in Background.\n- **Cleaner multicore verification.** A well-verified core reduces the multicore verification problem to coherence, consistency, interconnect, and shared resources. See `research/10-verification/multicore-verification.md`.\n- **Foundation for post-silicon.** First-stage silicon bring-up is dramatically simpler when the core is well-understood and the debug interfaces (assertions, traces, scan) are well-designed.\n- **Reusable testbench assets.** Generator, scoreboard, coverage model, and SVA library can be reused across cores if XH-1 has core variants, or across revisions of the same core.\n- **Formal evidence on critical control.** Privileged ISA semantics are hard to verify by random simulation; formal proofs on these blocks provide a high-confidence floor.\n\n## Disadvantages\n\nCosts and risks of a robust core verification program:\n\n- **Engineering cost.** A UVM-class environment plus formal plus emulation can easily exceed the RTL design cost for a single core.\n- **Tooling lock-in.** Heavy UVM usage often implies commercial simulator licensing; this may not match XH-1's tool posture.\n- **Slow turnaround.** Constrained-random simulation with full coverage at the core level is slow; coverage closure can take many cycles of \"add test, rerun, close hole\".\n- **Reference model dependence.** Spike, Sail, or any reference may diverge from the RISC-V specification in interpretation of ambiguous or profile-specific corners; the verification program must track the chosen RVA profile carefully and pin the spec version of the reference.\n- **Coverage false confidence.** Hitting a cover point is not the same as proving correctness; the verification plan must distinguish cover and check.\n- **128-core replication pressure.** If the core testbench is not designed to compose with the multicore testbench, redundant effort emerges.\n\n## XH-1 Considerations\n\nXH-1-specific factors that constrain the methodology choice:\n\n- **128-core replication.** The verification plan must be replication-aware: the *testbench* should be agnostic to whether one or 128 cores are instantiated. This implies strong separation of (a) the core DUT, (b) the core testbench, and (c) the multicore wrapper.\n- **Custom design.** No third-party core is being verified; there is no vendor's pre-built verification IP. All environments must be built or assembled.\n- **Repository context.** The related verification documents are all placeholders (\"SOON\"). This is a greenfield verification program; the XH-1 project has the opportunity — and the burden — to define the methodology from first principles.\n- **RISC-V ISA.** The RISC-V ecosystem provides mature reference models (Spike, Sail), compliance / architectural test frameworks (`riscv-tests`, `riscv-arch-test`), and open-source verification tooling (`riscv-formal`, `riscv-dv`). These are the natural anchors.\n- **Tool posture (OPEN QUESTION 3).** Without evidence in the repository, the project's tool posture (open-source only, commercial EDA, mixed) is OPEN. The methodology composition must remain valid across posture choices.\n- **Design language (OPEN).** The choice of design language — SystemVerilog, traditional Verilog, Chisel/Spinal HDL, Bluespec — has direct implications for verification methodology (e.g., formal property availability on generated RTL, simulation throughput, the place of unit-level testing at the HDL-source level vs. the generated-RTL level). This is OPEN pending XH-1 design-language disclosure.\n- **Microarchitectural class (OPEN QUESTION 2).** The pipeline class — in-order vs. out-of-order, superscalar width, depth of speculation, branch-prediction aggressiveness — materially changes the cost-benefit of formal vs. dynamic and the complexity of pipeline flush / replay verification. The methodology composition must accommodate both extremes.\n- **Power architecture (OPEN QUESTION 7).** Whether the core supports clock gating, power gating, retention, and/or DVFS is OPEN. The verification methodology must accommodate each combination. Assuming any specific power feature is invalid without evidence.\n\n## 128-Core Scalability\n\nReplication does not change which verification techniques apply to a single core, but it changes their cost-benefit and their composition.\n\n- **Core testbench scalability.** A core testbench written for a single instance must be instantiable inside a multicore wrapper without modification. This implies the testbench interfaces to the core through a standard transaction-level interface (or at least a clearly defined boundary), not through global signals.\n- **Cost-benefit of formal at replication.** Per-block formal verification has high fixed cost but a per-instance cost of zero: once a control property is proven, it holds for every replicated instance. Under the replication-homogeneity assumption, formal's per-die marginal cost decreases as the replication count increases. The opposite is true for dynamic simulation at full coverage: a per-instance cost must be paid for each replicated instance if all instances are to be exercised at full coverage. This trade-off is a primary argument for the hybrid approach at 128-core scale.\n- **Workload scaling.** Running the full core testbench in a 128-core environment is generally unnecessary; sampling (verify a subset of cores under full stimulus, sweep register state across all 128) is a more efficient pattern. The sampling strategy is a multicore concern but must be supported by the core testbench's reset/configuration interface. The statistical argument for the chosen sampling fraction (defect-escape probability vs. simulation cost) is a multicore-layer deliverable; the core testbench must support the interface.\n- **Coverage aggregation.** Per-core coverage data must be aggregated across the 128 instances. The core testbench should emit coverage in a form that is identifiable by instance index.\n- **Debug bandwidth.** 128 cores each generating waveform/log data is impractical. The core testbench must distinguish between detailed-trace mode (one core, full visibility) and lightweight-trace mode (128 cores, summary only).\n- **Compile and elaboration cost.** Elaboration of a 128-core testbench can dominate runtime. The core testbench should be elaborable standalone and in a multicore wrapper without recompiling the core RTL.\n- **Symmetry assumption revisit.** The replication-homogeneity assumption is load-bearing throughout this section. Heterogeneous cores (e.g., big/little, mixed application/accelerator, per-core ISA-subset configuration) invalidate the per-die marginal-cost argument and require per-variant verification, including per-variant testbenches, per-variant coverage, and per-variant formal. The methodology should be designed so that the additional cost of a variant is linear in the number of variants, not linear in 128 × variants.\n\n## Performance Considerations\n\nThe following are framework-level considerations. INSUFFICIENT EVIDENCE exists to provide numerical throughput estimates without XH-1-specific RTL data, simulator choice, and toolchain selection; any specific number is unsourced and is therefore not asserted here.\n\n- **Testbench throughput.** Dynamic simulation throughput determines how many random seeds can be run per unit wall-clock. Constrained-random at the core level is sensitive to simulator choice, RTL abstraction level, and coverage instrumentation; representative numbers are available in vendor documentation (commercial simulators) and in the open-source Verilator and cocotb benchmark literature, but are workload-dependent and cannot be quoted as typical without measurement on the XH-1 DUT.\n- **Compiler/optimizer.** Coverage instrumentation slows simulation. Production coverage runs and fast-turnaround debug runs should be separable.\n- **Co-simulation overhead.** Spike co-simulation overhead per committed instruction is non-trivial. Synchronization strategy (per-instruction vs. batched) is a tradeoff between debug granularity and throughput.\n- **FPGA prototype throughput.** FPGA prototype throughput is set by Fmax, which is determined by LUT count, routing, and the memory subsystem. Quoted MHz figures for an XH-1 core on a specific FPGA platform are not asserted here.\n- **Regression wall-clock budget.** A nightly regression that runs N seeds on a single core must fit within the available compute budget. Cloud burst capacity is typical.\n\nINSUFFICIENT EVIDENCE to provide numerical throughput estimates without XH-1-specific RTL data and toolchain selection.\n\n## Area Considerations\n\n- **On-die verification infrastructure.** XH-1 is described as a 128-core design. The verification *methodology* does not directly consume die area, but decisions made during verification (e.g., embedding assertions in RTL, adding observability muxes for trace, adding per-core DFT structures) do.\n- **Trace and debug infrastructure.** Core-level verification informs the design of trace, performance counters, and debug-transport blocks. Larger trace infrastructure costs area; the verification program must justify the cost in terms of debug productivity and post-silicon coverage.\n- **DFT (Design-for-Test) compatibility.** Scan chains, BIST, and MBIST are typically owned by a separate DFT program. Core verification should ensure that the core's DFT structures do not alter functional behavior in mission mode.\n- **Replication-driven area optimization.** If the core's area is a major constraint (which is likely at 128 cores on a single die), the verification plan must protect against defects that would force a re-spin with re-spinned core area.\n\nINSUFFICIENT EVIDENCE to quantify area without access to XH-1's core area budget or process node.\n\n## Power and Energy Considerations\n\nThe following considerations apply **if and only if** XH-1 implements the corresponding power features. Each is OPEN QUESTION 7-conditional.\n\n- **Power-aware verification (conditional).** If the core supports power gating or retention, the core testbench must be able to drive representative switching activity to enable power estimation, and the testbench must verify that power-gated domains behave correctly (controlled wake, retention restore, ordering with respect to coherence traffic).\n- **Energy proportionality (conditional).** If the core supports clock gating at the instruction or block level, the testbench should verify that idle and active transitions behave correctly at the architectural interface (no observable state loss in idle, correct wake on interrupt).\n- **Thermal and DVFS interactions (conditional).** If XH-1 supports DVFS at the core level, voltage/frequency transitions and their effects on memory coherence, pipeline flush behavior, and CSR access must be verified. This is a cross-cutting concern between core and system verification.\n- **Power estimation accuracy.** Pre-silicon power estimation depends on workload realism. A core testbench that only runs synthetic random code will produce misleading power numbers.\n\nNo power feature is assumed. Verification provisions for any power feature must be activated only when OPEN QUESTION 7 is resolved.\n\n## Implementation Considerations\n\n- **Methodology composition.** The project must select one of the alternative strategies (or a hybrid) and codify it in the verification plan.\n- **Toolchain selection.** Open-source (Verilator, cocotb, SymbiYosys, `riscv-formal`, `riscv-dv`, Spike, RISCV-ISAC, GTKWave) vs. commercial (Synopsys VCS, Cadence Xcelium, Siemens Questa, JasperGold, FormalPro) is a strategic decision. The choice cascades through licensing, hiring, training, and CI. These categories are not equivalent in role: Verilator and cocotb are simulation/cosim frameworks; SymbiYosys is a formal-verification flow; `riscv-formal` is an ISA-level formal property set; `riscv-dv` is a stimulus generator; RISCV-ISAC is a coverage tool. The choice within each role is independent.\n- **Reference model choice.** Spike (C++, instruction-accurate) vs. Sail (functional-language specification implementation) vs. a custom golden model. Spike is the most common reference; the project must document any corner-case behaviors of the chosen reference against the RISC-V specification.\n- **Coverage database.** Coverage data are stored in vendor-specific formats (e.g., Synopsys UCDB / DDC, Cadence IMC, Siemens UCDB) and may be aggregated into a custom backend (e.g., SQLite) for CI. The format is a tool-vendor concern; the aggregation backend is a CI concern. These are distinct choices.\n- **Test categorization.** Tests must be tagged by domain (base integer, M-ext, A-ext, F/D-ext, privileged, vector, …), by intent (sanity, corner, regression, performance), and by status (passing, known-fail, expected-fail).\n- **Seed management.** Random seeds should be reproducible. Seed pools should be partitioned into \"core regressions\", \"block regressions\", and \"fuzzing\".\n- **Bring-up sequencing.** Block-level → core-level → multicore → system. The core testbench is the transition point between block-level and system-level work.\n\n## Sign-Off Criteria\n\nA core verification sign-off is the point at which the methodology exits the coverage-closure funnel. The following are candidate criteria; numeric thresholds are PROPOSAL pending OPEN QUESTION resolution.\n\n- **Functional coverage.** All defined cover points hit, with cross-coverage exercised. Threshold PROPOSAL: ≥ 95% cover-point hit, ≥ 90% cross-coverage hit; remaining items must be justified as unreachable or formally proven. Final threshold to be set in the verification plan.\n- **Code coverage.** Statement, branch, expression, toggle, and FSM coverage reported; excluded regions justified. Threshold PROPOSAL: ≥ 95% statement, ≥ 90% branch, with manual exclusion of reset/DFT/clock-gating logic. Final threshold to be set in the verification plan.\n- **Assertion coverage.** All SVA / immediate assertions vacuously satisfied or actively proven; assertion cover points hit. Open assertions must be reviewed and dispositioned.\n- **Formal closure.** Per-block formal sign-off, with proof closure recorded for each property. Inconclusive results must be either strengthened (via abstraction / decomposition) or accepted with risk justification.\n- **ISA compliance evidence.** All applicable tests in `riscv-arch-test` (the architectural test framework) and `riscv-tests` (the legacy compliance suite) pass. Signed certificate or equivalent record retained.\n- **Mutation testing.** Periodic injection of known-defect mutations into a frozen RTL revision; a target mutation-kill rate PROPOSAL: ≥ 80% of injected mutations detected by the testbench + assertions + formal. Final threshold to be set in the verification plan.\n- **Reference model validation.** Spike or Sail validated against the chosen RISC-V compliance tests before being used as a checker; validation evidence archived.\n- **Testbench self-check verification.** Every scoreboard check has at least one negative test that is expected to fail.\n- **Reproducibility.** Every sign-off run is reproducible from committed source, seed, and tool version.\n\n## Verification Considerations\n\nCore verification is itself subject to verification — the verification of the verification environment:\n\n- **Testbench self-checking.** All scoreboard checks must be end-of-test verifiable; every check should be paired with at least one negative test that is expected to fail.\n- **Reference model validation.** The reference model (Spike, Sail) must be validated against the chosen RISC-V compliance tests before being used as a checker.\n- **Coverage sanity.** Cover points must be reviewed to ensure they are not trivially self-firing.\n- **Mutation testing.** A subset of RTL defects should be intentionally injected to confirm the testbench catches them. See `research/10-verification/fault-injection.md`.\n- **Tool qualification.** If the verification tools are used in any safety-relevant context, tool qualification (per ISO 26262, DO-254, or similar) is required. The applicability of these standards to XH-1 is OPEN QUESTION 8.\n- **Version control and baselines.** Every test run must be reproducible from committed source. Coverage baselines must be versioned.\n\n## Software Considerations\n\n- **Toolchain integration.** The core verification environment must integrate with the software toolchain (compiler, assembler, linker, Spike, debug server) to enable co-simulation and to run real programs.\n- **Firmware.** Firmware that runs on the core (boot code, exception handlers, interrupt dispatchers) is itself verification target; the core testbench should support loading and running firmware images.\n- **Operating system implications.** A core that supports an OS (Linux, RTOS) must be verified with syscall-heavy workloads. The core testbench should provide OS-aware run modes.\n- **Debug and trace protocols.** The core's debug-transport module (per the RISC-V Debug specification) must be exercised. JTAG or cJTAG interactions with the core's debug module are typically a core-level concern.\n- **Performance analysis tools.** Tools that analyze trace data, performance counters, and profile information should be able to consume the artifacts the core testbench produces.\n- **Software developer ergonomics.** The verification team's interfaces (test writers, coverage reviewers) should be ergonomic. A \"developer can run a single test and see results in under a minute\" goal is a useful operational target.\n\n## Security Considerations (Out of Scope, but Flagged)\n\n**ASSUMPTION (out-of-scope for this revision):** Transient-execution / side-channel verification (e.g., Spectre-class, Meltdown-class, and constant-time verification) is treated as out of scope for the XH-1 core verification layer for this revision. For a 128-core design with privilege boundaries, this is a real concern, and a follow-up document should cover it. The current methodology is sufficient for functional and microarchitectural correctness but does not, by itself, provide security assurance.\n\n## Proposed Direction\n\nPROPOSAL — A **hybrid methodology** for XH-1 core verification, with the following structure:\n\n1. **Dynamic simulation** as the primary verification engine, with the framework (UVM vs. lightweight) gated on OPEN QUESTION 3 (tool posture), OPEN QUESTION 4 (team size and timeline), and the design-language choice.\n - If UVM: pair with `riscv-dv` as the instruction-stream generator and Spike or Sail as the reference model.\n - If lightweight: pair `riscv-dv` (or a custom generator) with cocotb / Verilator and Spike.\n2. **Formal verification** applied to control-dominated blocks: fetch, decode, CSR file, MMU/TLB, exception/interrupt delivery, and the debug module. Coordinate with `research/10-verification/formal-verification.md` for tool selection. Apply `riscv-formal` as an ISA-level formal layer where the open-source flow is in scope.\n3. **Co-simulation against a RISC-V reference model** (Spike or Sail) for ISA-compliance evidence. Coordinate with `research/10-verification/isa-compliance.md`. Pin the RISC-V ISA, Privileged ISA, and Debug spec versions.\n4. **Directed tests** for XH-1 microarchitectural corner cases that random cannot reach economically. The corner cases to target are conditional on the in-order / out-of-order answer in OPEN QUESTION 2.\n5. **FPGA prototyping** for throughput-driven verification and as a substrate for fault injection and multicore bring-up.\n6. **Regression infrastructure** with version-controlled seeds, baselines, and coverage data. Coordinate with `research/10-verification/regression.md`.\n7. **Coverage closure** as an explicit, tracked funnel with the sign-off criteria in the Sign-Off Criteria section. Numeric thresholds are deferred to the verification plan.\n8. **Reusable testbench architecture** so the core testbench composes inside the multicore testbench without modification, supporting both the per-die cost argument and the sampling strategy.\n9. **Mutation testing** as a periodic check that the testbench is still doing useful work. Coordinate with `research/10-verification/fault-injection.md`.\n\nThe hybrid is proposed; the specific framework within the hybrid (UVM vs. lightweight) is gated on OPEN QUESTION 3 and OPEN QUESTION 4, and the power-feature sub-portion is gated on OPEN QUESTION 7. The document does not pick between alternatives that the OPEN QUESTIONS do not yet support.\n\n## Confidence\n\n- **High confidence:** The general methodology composition (dynamic + formal + co-simulation + directed + prototype + regression + coverage closure) corresponds to the IEEE 1800.2 UVM standard, the IEEE 1800 SystemVerilog assertion subset, and the RISC-V International published ISA / Privileged / Debug specifications, and is broadly accepted in industrial practice. The references supporting each component are cited at the point of use.\n- **High confidence:** The 128-core scaling argument (formal has near-zero per-instance marginal cost under replication homogeneity; dynamic at full coverage has per-instance cost) is a direct consequence of the replication-homogeneity assumption and does not require empirical data.\n- **Medium confidence:** Specific tool selections, sign-off numeric thresholds, and area/power tradeoffs, given the absence of repository evidence on the OPEN QUESTIONS.\n- **Low confidence (OPEN):** XH-1's tool licensing posture, core microarchitecture details, ISA profile selection, team size, design language, and power architecture. These drive the specifics within the proposed hybrid.\n\n## Open Questions\n\n1. What RISC-V profile does the XH-1 core implement? Candidate RISC-V profiles are defined against a base XLEN: e.g., RVA23U64 / RVA23S64 are 64-bit application / supervisor profiles, while a 32-bit core would follow a different profile track (e.g., RVA22U32). The reference model (Spike variant, Sail configuration) and the compliance test scope follow from this choice.\n2. Is the XH-1 core in-order or out-of-order? The verification plan complexity differs substantially. The instructions explicitly disallow assuming either without repository evidence.\n3. What is XH-1's toolchain posture (open-source only, commercial EDA, mixed)?\n4. What is the team size and timeline? Methodology cost scales with team capacity.\n5. What is the core's microarchitectural contract — what blocks are in scope for unit verification, and what is the boundary between core and multicore? See `research/10-verification/unit-verification.md` and `research/10-verification/multicore-verification.md`.\n6. Are the 128 cores architecturally identical, or are there variants (big/little, mixed application/accelerator, per-core ISA-subset configuration)? Heterogeneity changes the testbench architecture and invalidates the per-die cost-leverage argument.\n7. What is the core's power architecture — is clock gating, power gating, retention, or DVFS at the core level, and what are the state-retention semantics?\n8. What is XH-1's relationship to functional safety standards (ISO 26262, IEC 61508, DO-254, or none)?\n9. How will the XH-1 core's trace, debug, and performance-counter interfaces be designed, and how will core verification consume them?\n10. What is the expected defect rate target at tape-out, and how does the verification program demonstrate closure against that target?\n11. What is the XH-1 design language (SystemVerilog, Verilog-2001, Chisel/Spinal HDL, Bluespec), and does verification target the HDL source level, the generated-RTL level, or both?\n\n## Sources\n\n- IEEE 1800.2-2020, *Standard for Universal Verification Methodology Based on Accellera SystemVerilog 3.1*. Defines UVM.\n- IEEE 1800-2017, *Standard for SystemVerilog — Unified Hardware Design, Specification, and Verification Language*. Defines SVA and immediate assertions.\n- RISC-V International, *The RISC-V Instruction Set Manual, Volume I: Unprivileged Architecture* (versioned).\n- RISC-V International, *The RISC-V Instruction Set Manual, Volume II: Privileged Architecture* (versioned).\n- RISC-V International, *RISC-V External Debug Support* (versioned).\n- RISC-V International, *RISC-V Profiles* (versioned, e.g., RVA23).\n- Symbiotic EDA / YosysHQ, *SymbiYosys* documentation (open-source formal verification flow).\n- `riscv-formal` framework (Verilog, BSD-licensed), maintained at github.com/YosysHQ/riscv-formal.\n- `riscv-dv` instruction-stream generator, maintained at github.com/google/riscv-dv.\n- `riscv-tests` and `riscv-arch-test`, maintained by RISC-V International.\n- Spike (the official RISC-V ISA simulator), maintained at github.com/riscv-software-src/riscv-isa-sim.\n- Sail RISC-V formal specification, maintained at github.com/rems-project/sail.\n- Watson et al., *Kami: a Single-Document, Executable, Formal Specification of the RISC-V ISA* (Cambridge). Open-source formal ISA specification in Coq with a Bluespec front-end.\n\nQuantitative throughput, area, power, and timing claims are not asserted in the body. The XH-1 repository documents referenced (under `research/10-verification/`) are placeholders (\"SOON\") and do not provide factual content to cite. No measurements, benchmarks, processor capabilities, URLs outside the repositories named above, or experimental results have been invented.","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":9287,"completion_tokens":8633,"total_tokens":17920,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":135,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round3.md b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round3.md new file mode 100644 index 0000000..6cf0e1e --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round3.md @@ -0,0 +1,370 @@ +# Core Verification + +## Status + +DRAFT — Research framework established; all claims are explicitly classified by epistemic status (FACT with citation, ASSUMPTION, PROPOSAL, RECOMMENDATION, OPEN QUESTION, or INSUFFICIENT EVIDENCE). Concrete sign-off criteria, tool selection, and quantitative thresholds are deferred until the OPEN QUESTION inputs are resolved. Example numeric thresholds in the Sign-Off section are starting points for the verification plan and are not validated targets. + +## Abstract + +This document investigates verification strategies for the XH-1 core, defined here as a single processing core instance within a custom 128-core RISC-V processor. Core verification is the layer of the verification methodology stack concerned with demonstrating that one core conforms to its architectural specification, microarchitectural contract, and integration assumptions prior to and independent of multi-core interactions. For a 128-core design, the core is the unit of replication; defects undetected at this layer replicate 128-fold across the die, amplifying area, power, yield, and debug cost. This document frames the research questions, surveys existing approaches (simulation, formal verification, hardware-assisted verification, FPGA prototyping, emulation), analyzes their applicability to a custom RISC-V core, and identifies the gaps that the XH-1 project must close given that the existing related documents in `research/10-verification/` are placeholders ("SOON"). + +## Research Question + +What verification methodology, toolchain, environment abstraction, coverage model, and sign-off criteria are appropriate for verifying a single XH-1 core such that the methodology: + +1. Provides evidence that the core implements the RISC-V ISA subset selected for XH-1 (see `research/10-verification/isa-compliance.md`). +2. Provides evidence that the core's microarchitecture meets timing, area, and power targets for a design that will be replicated 128 times. +3. Scales economically — verification effort, compute cost, and human review time must not grow linearly with the 128-core replication factor. +4. Produces artifacts (testbenches, coverage models, formal proofs, regression suites) that compose cleanly with the surrounding verification layers (unit, multicore, system, post-silicon). +5. Is implementable given that the related XH-1 documents in area 10 are not yet populated, indicating a greenfield verification program. + +## Background + +Verification of a processor core is conventionally decomposed into: + +- **Unit verification** — verification of individual blocks (ALU, FPU, register file, branch predictor, fetch/decode structures, load-store unit, replay logic, CSR file, MMU/TLB, caches). See `research/10-verification/unit-verification.md`. +- **Core verification** — verification of the integrated core as a single design under test (DUT), exercising instruction-stream and exception behavior at the architectural interface, plus internal microarchitectural invariants. +- **Multicore verification** — coherence, consistency, interconnect, and shared-resource correctness across multiple core instances. See `research/10-verification/multicore-verification.md`. + +For XH-1, the core is the replication unit. The cost model is asymmetric under the replication-homogeneity assumption below: the marginal cost of catching a defect at the core level is lower than the marginal cost of catching it after tape-out, in silicon debug, or in the field. The phrase "every bug caught at the core layer is caught 128 times" is shorthand for: every defect that escapes core verification is replicated 128-fold in silicon, multiplying area, power, yield, and debug impact. The verification effort itself is paid once; the fix is paid once; the unfixed-defect cost is multiplied by the replication count. + +**ASSUMPTION (replication homogeneity):** The 128 XH-1 cores are architecturally identical instances of a single RTL core produced by elaboration of the same source. Heterogeneous cores, per-core configuration, or runtime variants invalidate the replication-leverage argument and require per-variant verification. This is treated as an assumption because no repository evidence contradicts it; OPEN QUESTION 6 below revisits it. Note that the formal "per-instance cost of zero" claim for proven properties holds for bit-identical replicated RTL; if the source is parameterized and elaborated with per-instance parameters, the proof must be re-established for any parameter set that differs, and the cost is not strictly zero. + +**ASSUMPTION (target implementation class):** XH-1 targets ASIC tape-out. FPGA-only deployment would change the methodology (no DFT/ATPG, different sign-off criteria, no post-silicon bring-up in the same sense). This assumption is conditional and is revisited in OPEN QUESTION 12. + +**ASSUMPTION (verification language availability):** The verification language of the testbench is the same as or interoperable with the design language. This is revisited in OPEN QUESTION 11. + +Core verification draws on several formal and semi-formal techniques: + +- **Dynamic simulation** with constrained-random or directed tests, against an ISA reference model, with functional coverage. +- **Static (formal) verification** — model checking (bounded model checking, k-induction, IC3/PDR), equivalence checking (combinational and sequential), and property proving against the RTL. +- **Hardware-assisted verification** — FPGA prototyping and emulation (commercial emulators, open-source flows such as FireSim on Amazon EC2 F1), trading compile time for runtime speed. +- **Co-simulation** — two simulation engines running in lockstep (e.g., an RTL simulator and an architectural reference-model engine such as Spike) synchronized at defined events. +- **Waveform and assertion-based methodologies** — SystemVerilog Assertions (SVA), PSL, custom monitors, end-of-test checkers. + +Coverage is a central concern. Functional coverage, code coverage (line, branch, expression, toggle, FSM), and cross-coverage of architectural state are the standard dimensions. For a custom ISA implementation, the relationship between ISA compliance coverage and microarchitectural coverage must be defined explicitly. + +Standard core-verification concerns that must be addressed in addition to instruction-stream behavior: + +- **Lint** — static analysis of RTL style, common bug patterns, and synthesizability (e.g., Verilator lint, commercial lint tools). A prerequisite to dynamic and formal work. +- **Clock-domain crossing (CDC) and reset-domain crossing (RDC)** — analysis of signals crossing asynchronous clock or reset domains. Typically owned by a separate methodology but the core must be CDC/RDC-clean for integration. +- **X-propagation** — controlled handling of unknown ('X) propagation in simulation, using X-pessimism removal or X-init optimization, to avoid false coverage from masked X's. +- **Reset verification** — power-on reset, software reset, debug reset, partial reset, and reset-domain interaction. +- **Power intent** — if power features are present (see OPEN QUESTION 7), UPF/CPF power intent formats integrate with simulation to model power-gated, retention, and isolation behavior. +- **DFT/ATPG interaction** — scan, BIST, and MBIST structures must not alter mission-mode behavior; their interaction with the core testbench is a coordination point. + +These are not optional; their absence in a verification plan is a known gap in pre-silicon methodology. + +## Existing Approaches + +The following approaches are established in industry and academic practice. Their applicability to XH-1 is analyzed; none is recommended in isolation. + +### A. Constrained-Random Simulation with UVM + +**FACT:** Universal Verification Methodology (UVM) is standardized as IEEE 1800.2-2020, *Standard for Universal Verification Methodology Based on Accellera SystemVerilog 3.1* [IEEE 1800.2-2020], and is widely used in industrial block- and core-level processor verification for constrained-random stimulus generation, reference model integration, self-checking scoreboards, functional coverage, and phasing. + +**ASSUMPTION (conditional on design language):** UVM is applicable only if the XH-1 design language supports SystemVerilog class constructs — either the source is SystemVerilog, or the verification environment targets SystemVerilog regardless of source language. If the design language is Verilog-2001 only, or if the project commits to a Chisel/Spinal HDL or Bluespec-only flow, UVM is not directly applicable at the HDL-source level and an alternative testbench harness is required. This assumption is revisited in OPEN QUESTION 11. + +**PROPOSAL (applicability to XH-1):** Apply UVM as the primary dynamic verification environment for XH-1 core-level testing, paired with a chosen ISA reference (Spike or Sail). Sequence generation over the RISC-V ISA state space; reference model in co-simulation; scoreboard for architectural state comparison per committed instruction; explicit cover points for instruction × operand-mode × privilege × exception interactions. + +### B. Directed and Hand-Written Tests + +**FACT:** Constrained-random alone does not reach corner-case coverage of complex pipelines, corner-case CSR interactions, or rare exception sequences. Industry practice supplements with directed tests written by architects and designers. + +**PROPOSAL (applicability to XH-1):** A XH-1-specific directed test suite for pipeline hazard combinations specific to the XH-1 microarchitecture, corner cases in the chosen RVA profile subset, reset, wake, low-power entry/exit, and debug transport (if present at core level). + +### C. Formal Property Verification + +**FACT:** Bounded model checking, k-induction, and IC3/PDR are applied to control-dominated blocks: fetch, decode, CSR, MMU/TLB, branch predictor, retirement, and exception delivery. Commercial model checkers include Cadence JasperGold, Synopsys VC Formal, and Siemens FormalPro. Open-source flows include SymbiYosys (SBY), based on the Yosys synthesis and formal toolchain. Equivalence checking is a distinct formal category: tools include Synopsys Formality and Cadence Conformal, used to verify that two RTL representations (e.g., pre- and post-synthesis, or pre- and post-retiming) are functionally equivalent. Equivalence checkers are not model checkers and are not used for property proving. + +**PROPOSAL (applicability to XH-1):** Per-block formal sign-off for the control-dominated blocks listed above, end-to-end pipeline flushing, exception priority, and CSR read/write semantics. Formal proofs of privileged ISA semantics (e.g., `mstatus`, `sstatus`, `satp`, `mip`/`sip` behavior) provide a high-confidence anchor that random tests cannot match in isolation. Sequential equivalence checking between major RTL revisions is a separate, complementary activity. + +**ASSUMPTION (RISC-V privileged-ISA versioning):** The CSR semantics used as formal-proof targets correspond to the specific RISC-V Privileged ISA version adopted by XH-1. The unprivileged ISA, privileged ISA, and Debug specification each have versioned releases tracked by RISC-V International; the verification plan must be pinned to those versions [RISC-V International, *Unprivileged Specification*, *Privileged Specification*, *Debug Specification*]. + +**PROPOSAL (applicability to XH-1):** Coordinate with `research/10-verification/formal-verification.md` for tool selection (commercial model checkers vs. SymbiYosys / SBY) and with `research/10-verification/isa-compliance.md` for the privileged-ISA version. + +### D. Co-Simulation with an ISA Reference + +**FACT:** Comparing DUT architectural state against Spike (the official RISC-V ISA simulator, C++ implementation) or Sail (a first-order imperative language designed for ISA specification, with a RISC-V model generated from the RISC-V ISA formal definition) per instruction is a widely deployed pattern. Spike + RTL constitutes co-simulation in the strict sense: two simulation engines synchronized at defined events. The DUT may run ahead of the reference using a deferred-check queue, with results compared at synchronization points. A single simulator that incorporates the reference model as a transactor is more accurately described as reference-model checking rather than co-simulation. + +**PROPOSAL (applicability to XH-1):** Required for ISA-compliance evidence. See `research/10-verification/isa-compliance.md`. Reference model parallelism is a 128-core scaling concern: Spike and Sail are single-threaded by default; running 128 reference models in lockstep with 128 RTL instances is itself a significant engineering problem. The pattern of choice is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling (multicore-layer decision) with the core testbench supporting the interface. + +### E. FPGA Prototyping and Emulation + +**FACT:** FPGA-based prototyping runs the RTL at higher throughput than software simulation, and emulation platforms (commercial: Palladium, Veloce, ZeBu; open-source: FireSim on Amazon EC2 F1) provide a middle ground. The throughput multiplier is workload- and instrumentation-dependent; published numbers for FPGA prototyping vs. software simulation range roughly from 10× to over 1000× depending on workload, Fmax, and observability overhead. Quoted multipliers are approximate and are not asserted here as specific numbers for XH-1. + +**PROPOSAL (applicability to XH-1):** Use FPGA prototyping to speed up core testbench bring-up by running entire test programs end-to-end, to enable fault-injection campaigns that are infeasible in simulation (see `research/10-verification/fault-injection.md`), and as the natural place to begin validating the multicore layer; core verification is the upstream prerequisite. Commercial emulation is a viable alternative for very large workloads or when observability requirements are relaxed; the decision depends on licensing posture (OPEN QUESTION 3) and workload characteristics. + +### F. Regression Infrastructure + +**FACT:** Regression management — version control of testbenches, seeds, baselines, coverage databases — is a load-bearing component of any non-trivial core verification program. See `research/10-verification/regression.md`. + +**PROPOSAL (applicability to XH-1):** The XH-1 regression infrastructure must support a single-core nightly run, a multicore run, and a system run with shared seeds where appropriate. + +### G. Coverage Closure Methodology + +**FACT:** Coverage closure combines functional coverage (defined by the verification plan), code coverage (uncovered RTL), and assertion coverage. Industry practice treats coverage closure as a managed funnel: cover points → cross coverage → corner case identification → directed/random test addition → formal where possible. Coverage baselines drift as RTL changes; a baseline-management policy is required. + +**PROPOSAL (applicability to XH-1):** Codify the funnel explicitly in the verification plan with tracked stages, baseline-management policy, and exit criteria per stage. + +### H. Assertion-Based Verification + +**FACT:** SystemVerilog Assertions (IEEE 1800-2017, *Standard for SystemVerilog*) and immediate assertions encode design intent and serve both as live checkers and as inputs to formal tools. + +**PROPOSAL (applicability to XH-1):** Define an SVA library for XH-1 invariants (e.g., "no retired instruction has a privileged side effect inconsistent with its mode", "no speculative architectural state is observable at the architectural interface"). + +### I. Open-Source ISA-Level Formal Verification + +**FACT:** The `riscv-formal` framework (YosysHQ / Symbiotic EDA, Verilog, BSD-licensed) provides a directly applicable open-source ISA-level formal verification flow for RISC-V cores, consisting of a checkable RISC-V ISA specification encoded as a set of formal properties that are conjoined with a wrapper around the DUT and discharged by an SMT solver. The Kami project (Cambridge) provides an alternative formal RISC-V ISA specification in Coq with a Bluespec front-end; the project documentation and the publication by the Kami team should be cited from the project's own canonical sources rather than from a secondary attribution. + +**PROPOSAL (applicability to XH-1):** Evaluate `riscv-formal` as a complement to block-level formal. It directly targets ISA-compliance evidence with open-source tooling and is particularly relevant given the greenfield status of the XH-1 verification program. + +### J. Open-Source Constrained-Random Generator + +**FACT:** Google's `riscv-dv` is an open-source UVM-based constrained-random instruction-stream generator for RISC-V, widely used in industrial and academic flows. The license should be confirmed against the current repository state at the time of toolchain selection; this document does not assert a specific license version. + +**PROPOSAL (applicability to XH-1):** Evaluate `riscv-dv` as a generator either within a UVM environment or as a stimulus source for a lightweight harness. + +## Alternative Designs + +The following alternative *strategies* for the core verification methodology itself are candidates for XH-1. They are not mutually exclusive; the project must select a composition. + +### Alternative 1: UVM-Centric with Formal Supplements + +- **Approach:** UVM as the workhorse for dynamic verification, with formal applied to control-dominated blocks. +- **Strengths:** Industry-standard tooling, well-understood coverage and sign-off conventions, mature ecosystem; integrates with `riscv-dv` as a generator and `riscv-formal` as an ISA-level formal layer. +- **Weaknesses:** UVM has a steep learning curve and high setup cost. For a 128-core design, the UVM environment must be designed for reuse to avoid 128× rewrite cost. Requires SystemVerilog (OPEN QUESTION 11). + +### Alternative 2: Lightweight Python/C++ Test Harness with Spike Co-Simulation + +- **Approach:** A generator produces instruction streams and a C++ testbench (e.g., based on cocotb, Verilator, or a custom framework) compares architectural state against Spike. +- **Strengths:** Lower setup cost than UVM. Excellent generator productivity. Easy integration with regression infrastructure. Open-source-friendly. `riscv-dv` can serve as the generator. Verilator-based flows are not bound to SystemVerilog UVM at the source level. +- **Weaknesses:** Less mature scoreboard/reuse infrastructure than UVM. Coverage closure tooling is less standardized. +- **Application to XH-1:** Reasonable choice if XH-1 has a small team, an open-source toolchain preference, or a constraint on commercial EDA licensing. + +### Alternative 3: Formal-First + +- **Approach:** Formal proofs form the primary verification evidence; dynamic simulation is used for reach and to validate properties that formal cannot discharge. +- **Strengths:** Highest assurance per proven property. Strong fit for control logic and CSR behavior. `riscv-formal` provides a directly applicable open-source ISA-level formal layer. Under the replication-homogeneity assumption, proven properties hold for all 128 instances at near-zero marginal cost (modulo parameter-set caveats in the Background assumption). +- **Weaknesses:** Pipeline formalization is hard. State-space complexity grows combinatorially for large cores. May not reach proof closure on timing-agnostic models without abstraction. Inconclusive results are not equivalent to failure; risk-acceptance must be tracked. + +### Alternative 4: Emulation-Centric + +- **Approach:** Skip large-scale software simulation; use FPGA/emulation throughput as the primary engine. +- **Strengths:** Excellent long-workload capability; native fault-injection support. +- **Weaknesses:** Poor waveform/observability; slow compile/turnaround; harder to drive corner-case stimulus; debug cycle is slow. + +### Alternative 5: Hybrid (PROPOSAL — See Proposed Direction) + +A combination of a dynamic framework (Alternative 1 or Alternative 2) for stimulus and reach, formal supplements (Alternative 3, including `riscv-formal` where applicable) for control logic, and FPGA prototyping/emulation for throughput and multicore bring-up. + +## Comparison + +The comparison is qualitative and is intended to make the trade-offs explicit, not to produce a numeric ranking. INSUFFICIENT EVIDENCE exists to rank absolute quality without XH-1-specific RTL data, tool licensing posture, and team size. The Hybrid column is a composition of the others and is not an independent point of comparison; the table is structured to show how each pure strategy performs on the dimensions, with the Hybrid column indicating how the strengths compose. + +| Dimension | UVM-Centric | Lightweight Harness | Formal-First | Emulation-Centric | Hybrid | +|---|---|---|---|---|---| +| Setup cost | High | Low | High | Medium | High | +| Per-test runtime | Slow | Slow–Medium | N/A (proof time) | Fast | Mixed | +| Corner-case reach | Strong (with seeds) | Strong (with fuzzing) | Strong (when discharged) | Weak–Medium | Strong | +| Assurance on control logic | Medium | Medium | High | Low–Medium | High | +| Debug ergonomics | Good (waveform, logs) | Good | Strong counterexamples | Weak | Mixed | +| Open-source viability | Medium (UVM IEEE 1800.2 is open; simulators are usually commercial) | Strong | Medium | Medium | Medium | +| Fit for 128-core reuse | High (if designed so) | High (if designed so) | High (under replication homogeneity) | High | High | +| Pre-tape-out risk reduction | Strong | Medium–Strong | Strong (partial; inconclusive ≠ proven) | Medium | Strong | + +## Advantages + +The advantages of investing in a robust core verification methodology for XH-1: + +- **Replication leverage.** Under the replication-homogeneity assumption, an unfixed defect in the core RTL is replicated 128-fold in silicon. The cost-leverage argument is about the cost of escaping defects, not about replicated verification effort: verification is performed once, the defect (if found) is fixed once, and the fixed RTL is then replicated. The benefit of core-level detection is the avoidance of multiplied post-silicon cost. +- **Cleaner multicore verification.** A well-verified core reduces the multicore verification problem to coherence, consistency, interconnect, and shared resources. See `research/10-verification/multicore-verification.md`. +- **Foundation for post-silicon.** First-stage silicon bring-up is dramatically simpler when the core is well-understood and the debug interfaces (assertions, traces, scan) are well-designed. Post-silicon validation feeds back into pre-silicon coverage targets; this feedback loop is not detailed in this document and is left to the post-silicon layer. +- **Reusable testbench assets.** Generator, scoreboard, coverage model, and SVA library can be reused across cores if XH-1 has core variants, or across revisions of the same core. +- **Formal evidence on critical control.** Privileged ISA semantics are hard to verify by random simulation; formal proofs on these blocks provide a high-confidence floor. + +## Disadvantages + +Costs and risks of a robust core verification program: + +- **Engineering cost.** A UVM-class environment plus formal plus emulation is a substantial program. The claim that it "can easily exceed the RTL design cost for a single core" is folkloric and is presented here as a planning consideration, not as a measured fact. +- **Tooling lock-in.** Heavy UVM usage often implies commercial simulator licensing; this may not match XH-1's tool posture. Formal tool lock-in is similar: commercial model checkers differ in property language and run scripts. +- **Slow turnaround.** Constrained-random simulation with full coverage at the core level is slow; coverage closure can take many cycles of "add test, rerun, close hole". Coverage baselines drift with RTL changes, requiring re-establishment. +- **Reference model dependence.** Spike, Sail, or any reference may diverge from the RISC-V specification in interpretation of ambiguous or profile-specific corners; the verification program must track the chosen RVA profile carefully and pin the spec version of the reference. +- **Coverage false confidence.** Hitting a cover point is not the same as proving correctness; the verification plan must distinguish cover and check. +- **128-core replication pressure.** If the core testbench is not designed to compose with the multicore testbench, redundant effort emerges. +- **Inconclusive formal ≠ proven.** Formal results of "inconclusive" must be risk-accepted, not treated as success. +- **Equivalent mutants.** Mutation-kill rates are affected by equivalent mutants (mutations that produce behaviorally identical RTL); kill-rate targets are therefore noisy and must be interpreted with caution. + +## XH-1 Considerations + +XH-1-specific factors that constrain the methodology choice: + +- **128-core replication.** The verification plan must be replication-aware: the *testbench* should be agnostic to whether one or 128 cores are instantiated. This implies strong separation of (a) the core DUT, (b) the core testbench, and (c) the multicore wrapper. +- **Custom design.** No third-party core is being verified; there is no vendor's pre-built verification IP. All environments must be built or assembled. +- **Repository context.** The related verification documents are all placeholders ("SOON"). This is a greenfield verification program; the XH-1 project has the opportunity — and the burden — to define the methodology from first principles. +- **RISC-V ISA.** The RISC-V ecosystem provides mature reference models (Spike, Sail), compliance / architectural test frameworks (`riscv-tests`, `riscv-arch-test`), and open-source verification tooling (`riscv-formal`, `riscv-dv`). These are the natural anchors. +- **Tool posture (OPEN QUESTION 3).** Without evidence in the repository, the project's tool posture (open-source only, commercial EDA, mixed) is OPEN. The methodology composition must remain valid across posture choices. +- **Design language (OPEN QUESTION 11).** The choice of design language — SystemVerilog, traditional Verilog, Chisel/Spinal HDL, Bluespec — has direct implications for verification methodology (e.g., formal property availability on generated RTL, simulation throughput, the place of unit-level testing at the HDL-source level vs. the generated-RTL level, and the applicability of UVM at the source level). This is OPEN pending XH-1 design-language disclosure. Section A's UVM assumption is conditional on this resolution. +- **Microarchitectural class (OPEN QUESTION 2).** The pipeline class — in-order vs. out-of-order, superscalar width, depth of speculation, branch-prediction aggressiveness — materially changes the cost-benefit of formal vs. dynamic and the complexity of pipeline flush / replay verification. The methodology composition must accommodate both extremes. +- **Power architecture (OPEN QUESTION 7).** Whether the core supports clock gating, power gating, retention, and/or DVFS is OPEN. The verification methodology must accommodate each combination. Assuming any specific power feature is invalid without evidence. Power intent (UPF/CPF) integration is conditional on this. +- **Target implementation class (OPEN QUESTION 12).** Whether XH-1 targets ASIC tape-out or FPGA-only deployment fundamentally changes the verification methodology (DFT/ATPG scope, post-silicon layer, sign-off criteria). Treated here as an assumption (ASIC) pending resolution. + +## 128-Core Scalability + +Replication does not change which verification techniques apply to a single core, but it changes their cost-benefit and their composition. + +- **Core testbench scalability.** A core testbench written for a single instance must be instantiable inside a multicore wrapper without modification. This implies the testbench interfaces to the core through a standard transaction-level interface (or at least a clearly defined boundary), not through global signals. +- **Cost-benefit of formal at replication.** Per-block formal verification has high fixed cost but, under bit-identical replication, a per-instance marginal cost of zero for the proven properties: once a control property is proven, it holds for every replicated instance of the same RTL. If the source is parameterized and elaborated with per-instance parameters, the proof must be re-established for any parameter set that differs. Under the replication-homogeneity assumption, formal's per-die marginal cost decreases as the replication count increases. The opposite is true for dynamic simulation at full coverage: a per-instance simulation cost must be paid for each replicated instance if all instances are to be exercised at full coverage, though the wall-clock growth is not strictly linear (the simulator elaborates and runs them in the same run, and the per-instance cost is amortized across the simulation kernel). The trade-off is a primary argument for the hybrid approach at 128-core scale, and the practical mitigation is sampling, not full per-instance coverage. +- **Workload scaling.** Running the full core testbench in a 128-core environment is generally unnecessary; sampling (verify a subset of cores under full stimulus, sweep register state across all 128) is a more efficient pattern. The sampling strategy is a multicore concern but must be supported by the core testbench's reset/configuration interface. The statistical argument for the chosen sampling fraction (defect-escape probability vs. simulation cost) is a multicore-layer deliverable; the core testbench must support the interface. +- **Coverage aggregation.** Per-core coverage data must be aggregated across the 128 instances. The core testbench should emit coverage in a form that is identifiable by instance index. +- **Debug bandwidth.** 128 cores each generating waveform/log data is impractical. The core testbench must distinguish between detailed-trace mode (one core, full visibility) and lightweight-trace mode (128 cores, summary only). +- **Compile and elaboration cost.** Elaboration of a 128-core testbench can dominate runtime. The core testbench should be elaborable standalone and in a multicore wrapper without recompiling the core RTL. Elaboration-time behavior (reset, clock tree, memory subsystem) can differ between 1-core and 128-core contexts; the core testbench must be valid in both. +- **Testbench memory footprint.** 128 instances of a complex scoreboard and 128 reference models can dominate host memory. Memory-aware testbench design (shared reference models, deferred-check queues, summary-only trace mode) is required. This is acknowledged here and detailed in the multicore layer. +- **Reference-model parallelism.** Spike and Sail are single-threaded by default. Running 128 reference models in lockstep with 128 RTL instances is a significant engineering problem. The recommended pattern is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling. This is a multicore-layer design constraint that the core testbench interface must support. +- **Symmetry assumption revisit.** The replication-homogeneity assumption is load-bearing throughout this section. Heterogeneous cores (e.g., big/little, mixed application/accelerator, per-core ISA-subset configuration) invalidate the per-die marginal-cost argument and require per-variant verification, including per-variant testbenches, per-variant coverage, and per-variant formal. The methodology should be designed so that the additional cost of a variant is linear in the number of variants, not linear in 128 × variants. + +## Performance Considerations + +The following are framework-level considerations. INSUFFICIENT EVIDENCE exists to provide numerical throughput estimates without XH-1-specific RTL data, simulator choice, and toolchain selection; any specific number is unsourced and is therefore not asserted here. + +- **Testbench throughput.** Dynamic simulation throughput determines how many random seeds can be run per unit wall-clock. Constrained-random at the core level is sensitive to simulator choice, RTL abstraction level, and coverage instrumentation; representative numbers are available in vendor documentation (commercial simulators) and in the open-source Verilator and cocotb benchmark literature, but are workload-dependent and cannot be quoted as typical without measurement on the XH-1 DUT. +- **Compiler/optimizer.** Coverage instrumentation slows simulation. Production coverage runs and fast-turnaround debug runs should be separable. +- **Co-simulation overhead.** Spike co-simulation overhead per committed instruction is non-trivial. Synchronization strategy (per-instruction vs. batched) is a tradeoff between debug granularity and throughput. At 128 instances, the cost of a per-instruction synchronization across all instances is prohibitive; batched or sampled synchronization is the practical pattern. +- **FPGA prototype throughput.** FPGA prototype throughput is set by Fmax, which is determined by LUT count, routing, and the memory subsystem. Quoted MHz figures for an XH-1 core on a specific FPGA platform are not asserted here. +- **Regression wall-clock budget.** A nightly regression that runs N seeds on a single core must fit within the available compute budget. Cloud burst capacity is typical. + +INSUFFICIENT EVIDENCE to provide numerical throughput estimates without XH-1-specific RTL data and toolchain selection. + +## Area Considerations + +- **On-die verification infrastructure.** XH-1 is described as a 128-core design. The verification *methodology* does not directly consume die area, but decisions made during verification (e.g., embedding assertions in RTL, adding observability muxes for trace, adding per-core DFT structures) do. +- **Trace and debug infrastructure.** Core-level verification informs the design of trace, performance counters, and debug-transport blocks. Larger trace infrastructure costs area; the verification program must justify the cost in terms of debug productivity and post-silicon coverage. +- **DFT (Design-for-Test) compatibility.** Scan chains, BIST, and MBIST are typically owned by a separate DFT program. Core verification should ensure that the core's DFT structures do not alter functional behavior in mission mode. If XH-1 is FPGA-only (OPEN QUESTION 12), DFT is out of scope. +- **Replication-driven area optimization.** If the core's area is a major constraint (which is likely at 128 cores on a single die), the verification plan must protect against defects that would force a re-spin with re-spinned core area. + +INSUFFICIENT EVIDENCE to quantify area without access to XH-1's core area budget or process node. + +## Power and Energy Considerations + +The following considerations apply **if and only if** XH-1 implements the corresponding power features. Each is OPEN QUESTION 7-conditional. + +- **Power-aware verification (conditional).** If the core supports power gating or retention, the core testbench must be able to drive representative switching activity to enable power estimation, and the testbench must verify that power-gated domains behave correctly (controlled wake, retention restore, ordering with respect to coherence traffic). UPF/CPF power intent integration with the simulator is the standard mechanism. +- **Energy proportionality (conditional).** If the core supports clock gating at the instruction or block level, the testbench should verify that idle and active transitions behave correctly at the architectural interface (no observable state loss in idle, correct wake on interrupt). +- **Thermal and DVFS interactions (conditional).** If XH-1 supports DVFS at the core level, voltage/frequency transitions and their effects on memory coherence, pipeline flush behavior, and CSR access must be verified. This is a cross-cutting concern between core and system verification. +- **Power estimation accuracy.** Pre-silicon power estimation depends on workload realism. A core testbench that only runs synthetic random code will produce misleading power numbers. + +No power feature is assumed. Verification provisions for any power feature must be activated only when OPEN QUESTION 7 is resolved. + +## Implementation Considerations + +- **Methodology composition.** The project must select one of the alternative strategies (or a hybrid) and codify it in the verification plan. +- **Toolchain selection.** Open-source (Verilator, cocotb, SymbiYosys, `riscv-formal`, `riscv-dv`, Spike, RISCV-ISAC, GTKWave) vs. commercial (Synopsys VCS, Cadence Xcelium, Siemens Questa, JasperGold, VC Formal, FormalPro, Formality, Conformal, Palladium, Veloce, ZeBu) is a strategic decision. The choice cascades through licensing, hiring, training, and CI. These categories are not equivalent in role: Verilator and cocotb are simulation/cosim frameworks; SymbiYosys is a formal-verification flow; `riscv-formal` is an ISA-level formal property set; `riscv-dv` is a stimulus generator; RISCV-ISAC is a coverage tool; commercial emulators are hardware-assisted verification platforms distinct from FPGA prototyping. The choice within each role is independent. +- **Reference model choice.** Spike (C++, instruction-accurate) vs. Sail (first-order imperative ISA-specification language with a RISC-V model) vs. a custom golden model. Spike is the most common reference; the project must document any corner-case behaviors of the chosen reference against the RISC-V specification. +- **Coverage database.** Coverage data are stored in vendor-specific formats (e.g., Synopsys UCDB / DDC, Cadence IMC, Siemens UCDB) and may be aggregated into a custom backend (e.g., SQLite) for CI. The format is a tool-vendor concern; the aggregation backend is a CI concern. These are distinct choices. +- **Test categorization.** Tests must be tagged by domain (base integer, M-ext, A-ext, F/D-ext, privileged, vector, …), by intent (sanity, corner, regression, performance), and by status (passing, known-fail, expected-fail). +- **Seed management.** Random seeds should be reproducible. Seed pools should be partitioned into "core regressions", "block regressions", and "fuzzing". +- **Bring-up sequencing.** Block-level → core-level → multicore → system. The core testbench is the transition point between block-level and system-level work. + +## Sign-Off Criteria + +A core verification sign-off is the point at which the methodology exits the coverage-closure funnel. The following are candidate criteria; numeric thresholds are starting examples to be refined in the verification plan and are not validated targets. The relationship to DRAFT status is explicit: the numbers below are illustrative starting points, not sign-off-ready thresholds. + +- **Functional coverage.** All defined cover points hit, with cross-coverage exercised. Threshold starting example: ≥ 95% cover-point hit, ≥ 90% cross-coverage hit; remaining items must be justified as unreachable or formally proven. Final threshold to be set in the verification plan. +- **Code coverage.** Statement, branch, expression, toggle, and FSM coverage reported; excluded regions justified. Threshold starting example: ≥ 95% statement, ≥ 90% branch, with manual exclusion of reset/DFT/clock-gating logic. Final threshold to be set in the verification plan. +- **Assertion coverage.** All SVA / immediate assertions vacuously satisfied or actively proven; assertion cover points hit. Open assertions must be reviewed and dispositioned. +- **Formal closure.** Per-block formal sign-off, with proof closure recorded for each property. Inconclusive results must be either strengthened (via abstraction / decomposition) or accepted with risk justification; "inconclusive" is not equivalent to "proven". +- **ISA compliance evidence.** All applicable tests in `riscv-arch-test` (the architectural test framework) and `riscv-tests` (the legacy compliance suite) pass. Signed certificate or equivalent record retained. +- **Mutation testing.** Periodic injection of known-defect mutations into a frozen RTL revision; mutation-kill rate as a starting example: ≥ 80% of injected mutations detected by the testbench + assertions + formal. This number is a starting point and requires empirical validation on the XH-1 DUT. Equivalent-mutant handling (mutations that produce behaviorally identical RTL) must be defined; kill-rate targets are noisy and must be interpreted with caution. Final threshold and operator set to be set in the verification plan. +- **Reference model validation.** Spike or Sail validated against the chosen RISC-V compliance tests before being used as a checker; validation evidence archived. +- **Testbench self-check verification.** Every scoreboard check has at least one negative test that is expected to fail. +- **Reproducibility.** Every sign-off run is reproducible from committed source, seed, and tool version. +- **CDC/RDC closure.** CDC and RDC analysis clean for all core-level crossing structures. +- **Lint clean.** Lint clean at the policy-defined severity level. +- **Reset verification.** All reset domains and reset sequences verified; reset recovery verified. + +## Verification Considerations + +Core verification is itself subject to verification — the verification of the verification environment: + +- **Testbench self-checking.** All scoreboard checks must be end-of-test verifiable; every check should be paired with at least one negative test that is expected to fail. +- **Reference model validation.** The reference model (Spike, Sail) must be validated against the chosen RISC-V compliance tests before being used as a checker. +- **Coverage sanity.** Cover points must be reviewed to ensure they are not trivially self-firing. +- **Mutation testing.** A subset of RTL defects should be intentionally injected to confirm the testbench catches them. See `research/10-verification/fault-injection.md`. Equivalent-mutant handling must be defined. +- **Coverage baseline stability.** Coverage baselines drift as RTL changes; a baseline-management policy (when a baseline is re-established, by whom, with what approval) is required. +- **Tool qualification.** If the verification tools are used in any safety-relevant context, tool qualification (per ISO 26262, IEC 61508, DO-254, or similar) is required. The applicability of these standards to XH-1 is OPEN QUESTION 8. Tool qualification is a verification-of-verification concern, not only a sign-off checkbox. +- **Version control and baselines.** Every test run must be reproducible from committed source. Coverage baselines must be versioned. +- **Testbench verification at 128-instance scale.** The testbench itself has bugs. Verification of the testbench (negative tests, mutation of the testbench, equivalence of testbench revisions) is part of the methodology. At 128 instances, the testbench's own correctness becomes a non-trivial verification problem and is acknowledged here as a multicore-layer concern that the core testbench interface must support. + +## Software Considerations + +- **Toolchain integration.** The core verification environment must integrate with the software toolchain (compiler, assembler, linker, Spike, debug server) to enable co-simulation and to run real programs. +- **Firmware.** Firmware that runs on the core (boot code, exception handlers, interrupt dispatchers) is itself verification target; the core testbench should support loading and running firmware images. +- **Operating system implications.** A core that supports an OS (Linux, RTOS) must be verified with syscall-heavy workloads. The core testbench should provide OS-aware run modes. +- **Debug and trace protocols.** The core's debug-transport module (per the RISC-V Debug specification) must be exercised. JTAG or cJTAG interactions with the core's debug module are typically a core-level concern. +- **Performance analysis tools.** Tools that analyze trace data, performance counters, and profile information should be able to consume the artifacts the core testbench produces. +- **Software developer ergonomics.** The verification team's interfaces (test writers, coverage reviewers) should be ergonomic. A "developer can run a single test and see results in under a minute" goal is a useful operational target. + +## Security Considerations (Out of Scope, but Flagged) + +**ASSUMPTION (out-of-scope for this revision):** Transient-execution / side-channel verification (e.g., Spectre-class, Meltdown-class, and constant-time verification) is treated as out of scope for the XH-1 core verification layer for this revision. For a 128-core design with privilege boundaries, this is a real and high-risk concern. This is flagged as a high-risk gap and as a candidate for a follow-up document. The current methodology is sufficient for functional and microarchitectural correctness but does not, by itself, provide security assurance. + +## Proposed Direction + +PROPOSAL — A **hybrid methodology** for XH-1 core verification, with the following structure: + +1. **Dynamic simulation** as the primary verification engine, with the framework (UVM vs. lightweight) gated on OPEN QUESTION 3 (tool posture), OPEN QUESTION 4 (team size and timeline), and the design-language choice (OPEN QUESTION 11). + - If UVM: pair with `riscv-dv` as the instruction-stream generator and Spike or Sail as the reference model. + - If lightweight: pair `riscv-dv` (or a custom generator) with cocotb / Verilator and Spike. +2. **Formal verification** applied to control-dominated blocks: fetch, decode, CSR file, MMU/TLB, exception/interrupt delivery, and the debug module. Coordinate with `research/10-verification/formal-verification.md` for tool selection. Apply `riscv-formal` as an ISA-level formal layer where the open-source flow is in scope. Sequential equivalence checking between major RTL revisions is a complementary activity. +3. **Co-simulation against a RISC-V reference model** (Spike or Sail) for ISA-compliance evidence. Coordinate with `research/10-verification/isa-compliance.md`. Pin the RISC-V ISA, Privileged ISA, and Debug spec versions. At 128 instances, use a shared reference model with a deferred-check queue or sampled synchronization. +4. **Directed tests** for XH-1 microarchitectural corner cases that random cannot reach economically. The corner cases to target are conditional on the in-order / out-of-order answer in OPEN QUESTION 2. +5. **FPGA prototyping** for throughput-driven verification and as a substrate for fault injection and multicore bring-up. Commercial emulation is a conditional alternative based on licensing posture (OPEN QUESTION 3) and workload. +6. **Regression infrastructure** with version-controlled seeds, baselines, and coverage data. Coordinate with `research/10-verification/regression.md`. +7. **Coverage closure** as an explicit, tracked funnel with the sign-off criteria in the Sign-Off Criteria section, including baseline-management policy. Numeric thresholds are deferred to the verification plan and the numbers in the Sign-Off section are starting examples, not validated targets. +8. **Reusable testbench architecture** so the core testbench composes inside the multicore testbench without modification, supporting both the per-die cost argument and the sampling strategy. The interface must support shared reference models, summary-only trace mode, and per-instance coverage identification. +9. **Mutation testing** as a periodic check that the testbench is still doing useful work, with explicit equivalent-mutant handling. Coordinate with `research/10-verification/fault-injection.md`. +10. **Standard core-verification adjuncts:** lint, CDC/RDC analysis, X-propagation handling, reset verification, and (conditional on OPEN QUESTION 7) UPF/CPF power-intent integration. + +The hybrid is proposed; the specific framework within the hybrid (UVM vs. lightweight) is gated on OPEN QUESTION 3, OPEN QUESTION 4, and OPEN QUESTION 11, and the power-feature sub-portion is gated on OPEN QUESTION 7. The implementation-class sub-portion (DFT/ATPG scope, post-silicon layer) is gated on OPEN QUESTION 12. The document does not pick between alternatives that the OPEN QUESTIONS do not yet support. + +## Confidence + +- **High confidence:** The general methodology composition (dynamic + formal + co-simulation + directed + prototype + regression + coverage closure, with the standard adjuncts: lint, CDC/RDC, X-propagation, reset) corresponds to the IEEE 1800.2 UVM standard, the IEEE 1800 SystemVerilog assertion subset, and the RISC-V International published ISA / Privileged / Debug specifications, and is broadly accepted in industrial practice. The references supporting each component are cited at the point of use. +- **High confidence (conditional):** The 128-core scaling argument (formal has near-zero per-instance marginal cost under bit-identical replication homogeneity; dynamic at full coverage has per-instance cost) is a direct consequence of the replication-homogeneity assumption and does not require empirical data. The parameter-set caveat (re-elaboration for differing parameters) is acknowledged. +- **Medium confidence:** Specific tool selections, sign-off numeric thresholds, and area/power tradeoffs, given the absence of repository evidence on the OPEN QUESTIONS. +- **Low confidence (OPEN):** XH-1's tool licensing posture, core microarchitecture details, ISA profile selection, team size, design language, power architecture, and target implementation class. These drive the specifics within the proposed hybrid. + +## Open Questions + +1. What RISC-V profile does the XH-1 core implement? Candidate RISC-V profiles are defined against a base XLEN: e.g., RVA23U64 / RVA23S64 are 64-bit application / supervisor profiles, while a 32-bit core would follow a different profile track (e.g., RVA22U32). The reference model (Spike variant, Sail configuration) and the compliance test scope follow from this choice. +2. Is the XH-1 core in-order or out-of-order? The verification plan complexity differs substantially. The instructions explicitly disallow assuming either without repository evidence. +3. What is XH-1's toolchain posture (open-source only, commercial EDA, mixed)? +4. What is the team size and timeline? Methodology cost scales with team capacity. +5. What is the core's microarchitectural contract — what blocks are in scope for unit verification, and what is the boundary between core and multicore? See `research/10-verification/unit-verification.md` and `research/10-verification/multicore-verification.md`. +6. Are the 128 cores architecturally identical, or are there variants (big/little, mixed application/accelerator, per-core ISA-subset configuration)? Heterogeneity changes the testbench architecture and invalidates the per-die cost-leverage argument. +7. What is the core's power architecture — is clock gating, power gating, retention, or DVFS at the core level, and what are the state-retention semantics? +8. What is XH-1's relationship to functional safety standards (ISO 26262, IEC 61508, DO-254, or none)? +9. How will the XH-1 core's trace, debug, and performance-counter interfaces be designed, and how will core verification consume them? +10. What is the expected defect rate target at tape-out, and how does the verification program demonstrate closure against that target? +11. What is the XH-1 design language (SystemVerilog, Verilog-2001, Chisel/Spinal HDL, Bluespec), and does verification target the HDL source level, the generated-RTL level, or both? +12. What is XH-1's target implementation class — ASIC tape-out, FPGA-only deployment, or both? This determines DFT/ATPG scope, post-silicon validation depth, and sign-off criteria. +13. What physical implementation tool flow (synthesis, PnR, timing) is envisioned, and what are the timing-closure constraints (target clock frequency, slack budget)? + +## Sources + +- IEEE 1800.2-2020, *Standard for Universal Verification Methodology Based on Accellera SystemVerilog 3.1*. Defines UVM. +- IEEE 1800-2017, *Standard for SystemVerilog — Unified Hardware Design, Specification, and Verification Language*. Defines SVA and immediate assertions. +- RISC-V International, *The RISC-V Instruction Set Manual, Volume I: Unprivileged Architecture* (versioned). +- RISC-V International, *The RISC-V Instruction Set Manual, Volume II: Privileged Architecture* (versioned). +- RISC-V International, *RISC-V External Debug Support* (versioned). +- RISC-V International, *RISC-V Profiles* (versioned, e.g., RVA23). +- Symbiotic EDA / YosysHQ, *SymbiYosys* documentation (open-source formal verification flow). +- `riscv-formal` framework (Verilog, BSD-licensed), maintained at github.com/YosysHQ/riscv-formal. +- `riscv-dv` instruction-stream generator, maintained at github.com/google/riscv-dv. +- `riscv-tests` and `riscv-arch-test`, maintained by RISC-V International. +- Spike (the official RISC-V ISA simulator), maintained at github.com/riscv-software-src/riscv-isa-sim. +- Sail RISC-V formal specification, maintained at github.com/rems-project/sail. Sail is a first-order imperative ISA-specification language; the RISC-V model is generated from the RISC-V ISA formal definition. +- Kami project (Cambridge), formal RISC-V ISA specification in Coq with a Bluespec front-end. Project publications and documentation should be cited from the project's own canonical sources. + +Quantitative throughput, area, power, and timing claims are not asserted in the body. The XH-1 repository documents referenced (under `research/10-verification/`) are placeholders ("SOON") and do not provide factual content to cite. No measurements, benchmarks, processor capabilities, URLs outside the repositories named above, or experimental results have been invented. diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round3.md.json b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round3.md.json new file mode 100644 index 0000000..c5b5d85 --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round3.md.json @@ -0,0 +1,285 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787739706-5JRlijYpTUyCneJp1Cyv","object":"chat.completion","created":1787739706,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"# Core Verification\n\n## Status\n\nDRAFT — Research framework established; all claims are explicitly classified by epistemic status (FACT with citation, ASSUMPTION, PROPOSAL, RECOMMENDATION, OPEN QUESTION, or INSUFFICIENT EVIDENCE). Concrete sign-off criteria, tool selection, and quantitative thresholds are deferred until the OPEN QUESTION inputs are resolved. Example numeric thresholds in the Sign-Off section are starting points for the verification plan and are not validated targets.\n\n## Abstract\n\nThis document investigates verification strategies for the XH-1 core, defined here as a single processing core instance within a custom 128-core RISC-V processor. Core verification is the layer of the verification methodology stack concerned with demonstrating that one core conforms to its architectural specification, microarchitectural contract, and integration assumptions prior to and independent of multi-core interactions. For a 128-core design, the core is the unit of replication; defects undetected at this layer replicate 128-fold across the die, amplifying area, power, yield, and debug cost. This document frames the research questions, surveys existing approaches (simulation, formal verification, hardware-assisted verification, FPGA prototyping, emulation), analyzes their applicability to a custom RISC-V core, and identifies the gaps that the XH-1 project must close given that the existing related documents in `research/10-verification/` are placeholders (\"SOON\").\n\n## Research Question\n\nWhat verification methodology, toolchain, environment abstraction, coverage model, and sign-off criteria are appropriate for verifying a single XH-1 core such that the methodology:\n\n1. Provides evidence that the core implements the RISC-V ISA subset selected for XH-1 (see `research/10-verification/isa-compliance.md`).\n2. Provides evidence that the core's microarchitecture meets timing, area, and power targets for a design that will be replicated 128 times.\n3. Scales economically — verification effort, compute cost, and human review time must not grow linearly with the 128-core replication factor.\n4. Produces artifacts (testbenches, coverage models, formal proofs, regression suites) that compose cleanly with the surrounding verification layers (unit, multicore, system, post-silicon).\n5. Is implementable given that the related XH-1 documents in area 10 are not yet populated, indicating a greenfield verification program.\n\n## Background\n\nVerification of a processor core is conventionally decomposed into:\n\n- **Unit verification** — verification of individual blocks (ALU, FPU, register file, branch predictor, fetch/decode structures, load-store unit, replay logic, CSR file, MMU/TLB, caches). See `research/10-verification/unit-verification.md`.\n- **Core verification** — verification of the integrated core as a single design under test (DUT), exercising instruction-stream and exception behavior at the architectural interface, plus internal microarchitectural invariants.\n- **Multicore verification** — coherence, consistency, interconnect, and shared-resource correctness across multiple core instances. See `research/10-verification/multicore-verification.md`.\n\nFor XH-1, the core is the replication unit. The cost model is asymmetric under the replication-homogeneity assumption below: the marginal cost of catching a defect at the core level is lower than the marginal cost of catching it after tape-out, in silicon debug, or in the field. The phrase \"every bug caught at the core layer is caught 128 times\" is shorthand for: every defect that escapes core verification is replicated 128-fold in silicon, multiplying area, power, yield, and debug impact. The verification effort itself is paid once; the fix is paid once; the unfixed-defect cost is multiplied by the replication count.\n\n**ASSUMPTION (replication homogeneity):** The 128 XH-1 cores are architecturally identical instances of a single RTL core produced by elaboration of the same source. Heterogeneous cores, per-core configuration, or runtime variants invalidate the replication-leverage argument and require per-variant verification. This is treated as an assumption because no repository evidence contradicts it; OPEN QUESTION 6 below revisits it. Note that the formal \"per-instance cost of zero\" claim for proven properties holds for bit-identical replicated RTL; if the source is parameterized and elaborated with per-instance parameters, the proof must be re-established for any parameter set that differs, and the cost is not strictly zero.\n\n**ASSUMPTION (target implementation class):** XH-1 targets ASIC tape-out. FPGA-only deployment would change the methodology (no DFT/ATPG, different sign-off criteria, no post-silicon bring-up in the same sense). This assumption is conditional and is revisited in OPEN QUESTION 12.\n\n**ASSUMPTION (verification language availability):** The verification language of the testbench is the same as or interoperable with the design language. This is revisited in OPEN QUESTION 11.\n\nCore verification draws on several formal and semi-formal techniques:\n\n- **Dynamic simulation** with constrained-random or directed tests, against an ISA reference model, with functional coverage.\n- **Static (formal) verification** — model checking (bounded model checking, k-induction, IC3/PDR), equivalence checking (combinational and sequential), and property proving against the RTL.\n- **Hardware-assisted verification** — FPGA prototyping and emulation (commercial emulators, open-source flows such as FireSim on Amazon EC2 F1), trading compile time for runtime speed.\n- **Co-simulation** — two simulation engines running in lockstep (e.g., an RTL simulator and an architectural reference-model engine such as Spike) synchronized at defined events.\n- **Waveform and assertion-based methodologies** — SystemVerilog Assertions (SVA), PSL, custom monitors, end-of-test checkers.\n\nCoverage is a central concern. Functional coverage, code coverage (line, branch, expression, toggle, FSM), and cross-coverage of architectural state are the standard dimensions. For a custom ISA implementation, the relationship between ISA compliance coverage and microarchitectural coverage must be defined explicitly.\n\nStandard core-verification concerns that must be addressed in addition to instruction-stream behavior:\n\n- **Lint** — static analysis of RTL style, common bug patterns, and synthesizability (e.g., Verilator lint, commercial lint tools). A prerequisite to dynamic and formal work.\n- **Clock-domain crossing (CDC) and reset-domain crossing (RDC)** — analysis of signals crossing asynchronous clock or reset domains. Typically owned by a separate methodology but the core must be CDC/RDC-clean for integration.\n- **X-propagation** — controlled handling of unknown ('X) propagation in simulation, using X-pessimism removal or X-init optimization, to avoid false coverage from masked X's.\n- **Reset verification** — power-on reset, software reset, debug reset, partial reset, and reset-domain interaction.\n- **Power intent** — if power features are present (see OPEN QUESTION 7), UPF/CPF power intent formats integrate with simulation to model power-gated, retention, and isolation behavior.\n- **DFT/ATPG interaction** — scan, BIST, and MBIST structures must not alter mission-mode behavior; their interaction with the core testbench is a coordination point.\n\nThese are not optional; their absence in a verification plan is a known gap in pre-silicon methodology.\n\n## Existing Approaches\n\nThe following approaches are established in industry and academic practice. Their applicability to XH-1 is analyzed; none is recommended in isolation.\n\n### A. Constrained-Random Simulation with UVM\n\n**FACT:** Universal Verification Methodology (UVM) is standardized as IEEE 1800.2-2020, *Standard for Universal Verification Methodology Based on Accellera SystemVerilog 3.1* [IEEE 1800.2-2020], and is widely used in industrial block- and core-level processor verification for constrained-random stimulus generation, reference model integration, self-checking scoreboards, functional coverage, and phasing.\n\n**ASSUMPTION (conditional on design language):** UVM is applicable only if the XH-1 design language supports SystemVerilog class constructs — either the source is SystemVerilog, or the verification environment targets SystemVerilog regardless of source language. If the design language is Verilog-2001 only, or if the project commits to a Chisel/Spinal HDL or Bluespec-only flow, UVM is not directly applicable at the HDL-source level and an alternative testbench harness is required. This assumption is revisited in OPEN QUESTION 11.\n\n**PROPOSAL (applicability to XH-1):** Apply UVM as the primary dynamic verification environment for XH-1 core-level testing, paired with a chosen ISA reference (Spike or Sail). Sequence generation over the RISC-V ISA state space; reference model in co-simulation; scoreboard for architectural state comparison per committed instruction; explicit cover points for instruction × operand-mode × privilege × exception interactions.\n\n### B. Directed and Hand-Written Tests\n\n**FACT:** Constrained-random alone does not reach corner-case coverage of complex pipelines, corner-case CSR interactions, or rare exception sequences. Industry practice supplements with directed tests written by architects and designers.\n\n**PROPOSAL (applicability to XH-1):** A XH-1-specific directed test suite for pipeline hazard combinations specific to the XH-1 microarchitecture, corner cases in the chosen RVA profile subset, reset, wake, low-power entry/exit, and debug transport (if present at core level).\n\n### C. Formal Property Verification\n\n**FACT:** Bounded model checking, k-induction, and IC3/PDR are applied to control-dominated blocks: fetch, decode, CSR, MMU/TLB, branch predictor, retirement, and exception delivery. Commercial model checkers include Cadence JasperGold, Synopsys VC Formal, and Siemens FormalPro. Open-source flows include SymbiYosys (SBY), based on the Yosys synthesis and formal toolchain. Equivalence checking is a distinct formal category: tools include Synopsys Formality and Cadence Conformal, used to verify that two RTL representations (e.g., pre- and post-synthesis, or pre- and post-retiming) are functionally equivalent. Equivalence checkers are not model checkers and are not used for property proving.\n\n**PROPOSAL (applicability to XH-1):** Per-block formal sign-off for the control-dominated blocks listed above, end-to-end pipeline flushing, exception priority, and CSR read/write semantics. Formal proofs of privileged ISA semantics (e.g., `mstatus`, `sstatus`, `satp`, `mip`/`sip` behavior) provide a high-confidence anchor that random tests cannot match in isolation. Sequential equivalence checking between major RTL revisions is a separate, complementary activity.\n\n**ASSUMPTION (RISC-V privileged-ISA versioning):** The CSR semantics used as formal-proof targets correspond to the specific RISC-V Privileged ISA version adopted by XH-1. The unprivileged ISA, privileged ISA, and Debug specification each have versioned releases tracked by RISC-V International; the verification plan must be pinned to those versions [RISC-V International, *Unprivileged Specification*, *Privileged Specification*, *Debug Specification*].\n\n**PROPOSAL (applicability to XH-1):** Coordinate with `research/10-verification/formal-verification.md` for tool selection (commercial model checkers vs. SymbiYosys / SBY) and with `research/10-verification/isa-compliance.md` for the privileged-ISA version.\n\n### D. Co-Simulation with an ISA Reference\n\n**FACT:** Comparing DUT architectural state against Spike (the official RISC-V ISA simulator, C++ implementation) or Sail (a first-order imperative language designed for ISA specification, with a RISC-V model generated from the RISC-V ISA formal definition) per instruction is a widely deployed pattern. Spike + RTL constitutes co-simulation in the strict sense: two simulation engines synchronized at defined events. The DUT may run ahead of the reference using a deferred-check queue, with results compared at synchronization points. A single simulator that incorporates the reference model as a transactor is more accurately described as reference-model checking rather than co-simulation.\n\n**PROPOSAL (applicability to XH-1):** Required for ISA-compliance evidence. See `research/10-verification/isa-compliance.md`. Reference model parallelism is a 128-core scaling concern: Spike and Sail are single-threaded by default; running 128 reference models in lockstep with 128 RTL instances is itself a significant engineering problem. The pattern of choice is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling (multicore-layer decision) with the core testbench supporting the interface.\n\n### E. FPGA Prototyping and Emulation\n\n**FACT:** FPGA-based prototyping runs the RTL at higher throughput than software simulation, and emulation platforms (commercial: Palladium, Veloce, ZeBu; open-source: FireSim on Amazon EC2 F1) provide a middle ground. The throughput multiplier is workload- and instrumentation-dependent; published numbers for FPGA prototyping vs. software simulation range roughly from 10× to over 1000× depending on workload, Fmax, and observability overhead. Quoted multipliers are approximate and are not asserted here as specific numbers for XH-1.\n\n**PROPOSAL (applicability to XH-1):** Use FPGA prototyping to speed up core testbench bring-up by running entire test programs end-to-end, to enable fault-injection campaigns that are infeasible in simulation (see `research/10-verification/fault-injection.md`), and as the natural place to begin validating the multicore layer; core verification is the upstream prerequisite. Commercial emulation is a viable alternative for very large workloads or when observability requirements are relaxed; the decision depends on licensing posture (OPEN QUESTION 3) and workload characteristics.\n\n### F. Regression Infrastructure\n\n**FACT:** Regression management — version control of testbenches, seeds, baselines, coverage databases — is a load-bearing component of any non-trivial core verification program. See `research/10-verification/regression.md`.\n\n**PROPOSAL (applicability to XH-1):** The XH-1 regression infrastructure must support a single-core nightly run, a multicore run, and a system run with shared seeds where appropriate.\n\n### G. Coverage Closure Methodology\n\n**FACT:** Coverage closure combines functional coverage (defined by the verification plan), code coverage (uncovered RTL), and assertion coverage. Industry practice treats coverage closure as a managed funnel: cover points → cross coverage → corner case identification → directed/random test addition → formal where possible. Coverage baselines drift as RTL changes; a baseline-management policy is required.\n\n**PROPOSAL (applicability to XH-1):** Codify the funnel explicitly in the verification plan with tracked stages, baseline-management policy, and exit criteria per stage.\n\n### H. Assertion-Based Verification\n\n**FACT:** SystemVerilog Assertions (IEEE 1800-2017, *Standard for SystemVerilog*) and immediate assertions encode design intent and serve both as live checkers and as inputs to formal tools.\n\n**PROPOSAL (applicability to XH-1):** Define an SVA library for XH-1 invariants (e.g., \"no retired instruction has a privileged side effect inconsistent with its mode\", \"no speculative architectural state is observable at the architectural interface\").\n\n### I. Open-Source ISA-Level Formal Verification\n\n**FACT:** The `riscv-formal` framework (YosysHQ / Symbiotic EDA, Verilog, BSD-licensed) provides a directly applicable open-source ISA-level formal verification flow for RISC-V cores, consisting of a checkable RISC-V ISA specification encoded as a set of formal properties that are conjoined with a wrapper around the DUT and discharged by an SMT solver. The Kami project (Cambridge) provides an alternative formal RISC-V ISA specification in Coq with a Bluespec front-end; the project documentation and the publication by the Kami team should be cited from the project's own canonical sources rather than from a secondary attribution.\n\n**PROPOSAL (applicability to XH-1):** Evaluate `riscv-formal` as a complement to block-level formal. It directly targets ISA-compliance evidence with open-source tooling and is particularly relevant given the greenfield status of the XH-1 verification program.\n\n### J. Open-Source Constrained-Random Generator\n\n**FACT:** Google's `riscv-dv` is an open-source UVM-based constrained-random instruction-stream generator for RISC-V, widely used in industrial and academic flows. The license should be confirmed against the current repository state at the time of toolchain selection; this document does not assert a specific license version.\n\n**PROPOSAL (applicability to XH-1):** Evaluate `riscv-dv` as a generator either within a UVM environment or as a stimulus source for a lightweight harness.\n\n## Alternative Designs\n\nThe following alternative *strategies* for the core verification methodology itself are candidates for XH-1. They are not mutually exclusive; the project must select a composition.\n\n### Alternative 1: UVM-Centric with Formal Supplements\n\n- **Approach:** UVM as the workhorse for dynamic verification, with formal applied to control-dominated blocks.\n- **Strengths:** Industry-standard tooling, well-understood coverage and sign-off conventions, mature ecosystem; integrates with `riscv-dv` as a generator and `riscv-formal` as an ISA-level formal layer.\n- **Weaknesses:** UVM has a steep learning curve and high setup cost. For a 128-core design, the UVM environment must be designed for reuse to avoid 128× rewrite cost. Requires SystemVerilog (OPEN QUESTION 11).\n\n### Alternative 2: Lightweight Python/C++ Test Harness with Spike Co-Simulation\n\n- **Approach:** A generator produces instruction streams and a C++ testbench (e.g., based on cocotb, Verilator, or a custom framework) compares architectural state against Spike.\n- **Strengths:** Lower setup cost than UVM. Excellent generator productivity. Easy integration with regression infrastructure. Open-source-friendly. `riscv-dv` can serve as the generator. Verilator-based flows are not bound to SystemVerilog UVM at the source level.\n- **Weaknesses:** Less mature scoreboard/reuse infrastructure than UVM. Coverage closure tooling is less standardized.\n- **Application to XH-1:** Reasonable choice if XH-1 has a small team, an open-source toolchain preference, or a constraint on commercial EDA licensing.\n\n### Alternative 3: Formal-First\n\n- **Approach:** Formal proofs form the primary verification evidence; dynamic simulation is used for reach and to validate properties that formal cannot discharge.\n- **Strengths:** Highest assurance per proven property. Strong fit for control logic and CSR behavior. `riscv-formal` provides a directly applicable open-source ISA-level formal layer. Under the replication-homogeneity assumption, proven properties hold for all 128 instances at near-zero marginal cost (modulo parameter-set caveats in the Background assumption).\n- **Weaknesses:** Pipeline formalization is hard. State-space complexity grows combinatorially for large cores. May not reach proof closure on timing-agnostic models without abstraction. Inconclusive results are not equivalent to failure; risk-acceptance must be tracked.\n\n### Alternative 4: Emulation-Centric\n\n- **Approach:** Skip large-scale software simulation; use FPGA/emulation throughput as the primary engine.\n- **Strengths:** Excellent long-workload capability; native fault-injection support.\n- **Weaknesses:** Poor waveform/observability; slow compile/turnaround; harder to drive corner-case stimulus; debug cycle is slow.\n\n### Alternative 5: Hybrid (PROPOSAL — See Proposed Direction)\n\nA combination of a dynamic framework (Alternative 1 or Alternative 2) for stimulus and reach, formal supplements (Alternative 3, including `riscv-formal` where applicable) for control logic, and FPGA prototyping/emulation for throughput and multicore bring-up.\n\n## Comparison\n\nThe comparison is qualitative and is intended to make the trade-offs explicit, not to produce a numeric ranking. INSUFFICIENT EVIDENCE exists to rank absolute quality without XH-1-specific RTL data, tool licensing posture, and team size. The Hybrid column is a composition of the others and is not an independent point of comparison; the table is structured to show how each pure strategy performs on the dimensions, with the Hybrid column indicating how the strengths compose.\n\n| Dimension | UVM-Centric | Lightweight Harness | Formal-First | Emulation-Centric | Hybrid |\n|---|---|---|---|---|---|\n| Setup cost | High | Low | High | Medium | High |\n| Per-test runtime | Slow | Slow–Medium | N/A (proof time) | Fast | Mixed |\n| Corner-case reach | Strong (with seeds) | Strong (with fuzzing) | Strong (when discharged) | Weak–Medium | Strong |\n| Assurance on control logic | Medium | Medium | High | Low–Medium | High |\n| Debug ergonomics | Good (waveform, logs) | Good | Strong counterexamples | Weak | Mixed |\n| Open-source viability | Medium (UVM IEEE 1800.2 is open; simulators are usually commercial) | Strong | Medium | Medium | Medium |\n| Fit for 128-core reuse | High (if designed so) | High (if designed so) | High (under replication homogeneity) | High | High |\n| Pre-tape-out risk reduction | Strong | Medium–Strong | Strong (partial; inconclusive ≠ proven) | Medium | Strong |\n\n## Advantages\n\nThe advantages of investing in a robust core verification methodology for XH-1:\n\n- **Replication leverage.** Under the replication-homogeneity assumption, an unfixed defect in the core RTL is replicated 128-fold in silicon. The cost-leverage argument is about the cost of escaping defects, not about replicated verification effort: verification is performed once, the defect (if found) is fixed once, and the fixed RTL is then replicated. The benefit of core-level detection is the avoidance of multiplied post-silicon cost.\n- **Cleaner multicore verification.** A well-verified core reduces the multicore verification problem to coherence, consistency, interconnect, and shared resources. See `research/10-verification/multicore-verification.md`.\n- **Foundation for post-silicon.** First-stage silicon bring-up is dramatically simpler when the core is well-understood and the debug interfaces (assertions, traces, scan) are well-designed. Post-silicon validation feeds back into pre-silicon coverage targets; this feedback loop is not detailed in this document and is left to the post-silicon layer.\n- **Reusable testbench assets.** Generator, scoreboard, coverage model, and SVA library can be reused across cores if XH-1 has core variants, or across revisions of the same core.\n- **Formal evidence on critical control.** Privileged ISA semantics are hard to verify by random simulation; formal proofs on these blocks provide a high-confidence floor.\n\n## Disadvantages\n\nCosts and risks of a robust core verification program:\n\n- **Engineering cost.** A UVM-class environment plus formal plus emulation is a substantial program. The claim that it \"can easily exceed the RTL design cost for a single core\" is folkloric and is presented here as a planning consideration, not as a measured fact.\n- **Tooling lock-in.** Heavy UVM usage often implies commercial simulator licensing; this may not match XH-1's tool posture. Formal tool lock-in is similar: commercial model checkers differ in property language and run scripts.\n- **Slow turnaround.** Constrained-random simulation with full coverage at the core level is slow; coverage closure can take many cycles of \"add test, rerun, close hole\". Coverage baselines drift with RTL changes, requiring re-establishment.\n- **Reference model dependence.** Spike, Sail, or any reference may diverge from the RISC-V specification in interpretation of ambiguous or profile-specific corners; the verification program must track the chosen RVA profile carefully and pin the spec version of the reference.\n- **Coverage false confidence.** Hitting a cover point is not the same as proving correctness; the verification plan must distinguish cover and check.\n- **128-core replication pressure.** If the core testbench is not designed to compose with the multicore testbench, redundant effort emerges.\n- **Inconclusive formal ≠ proven.** Formal results of \"inconclusive\" must be risk-accepted, not treated as success.\n- **Equivalent mutants.** Mutation-kill rates are affected by equivalent mutants (mutations that produce behaviorally identical RTL); kill-rate targets are therefore noisy and must be interpreted with caution.\n\n## XH-1 Considerations\n\nXH-1-specific factors that constrain the methodology choice:\n\n- **128-core replication.** The verification plan must be replication-aware: the *testbench* should be agnostic to whether one or 128 cores are instantiated. This implies strong separation of (a) the core DUT, (b) the core testbench, and (c) the multicore wrapper.\n- **Custom design.** No third-party core is being verified; there is no vendor's pre-built verification IP. All environments must be built or assembled.\n- **Repository context.** The related verification documents are all placeholders (\"SOON\"). This is a greenfield verification program; the XH-1 project has the opportunity — and the burden — to define the methodology from first principles.\n- **RISC-V ISA.** The RISC-V ecosystem provides mature reference models (Spike, Sail), compliance / architectural test frameworks (`riscv-tests`, `riscv-arch-test`), and open-source verification tooling (`riscv-formal`, `riscv-dv`). These are the natural anchors.\n- **Tool posture (OPEN QUESTION 3).** Without evidence in the repository, the project's tool posture (open-source only, commercial EDA, mixed) is OPEN. The methodology composition must remain valid across posture choices.\n- **Design language (OPEN QUESTION 11).** The choice of design language — SystemVerilog, traditional Verilog, Chisel/Spinal HDL, Bluespec — has direct implications for verification methodology (e.g., formal property availability on generated RTL, simulation throughput, the place of unit-level testing at the HDL-source level vs. the generated-RTL level, and the applicability of UVM at the source level). This is OPEN pending XH-1 design-language disclosure. Section A's UVM assumption is conditional on this resolution.\n- **Microarchitectural class (OPEN QUESTION 2).** The pipeline class — in-order vs. out-of-order, superscalar width, depth of speculation, branch-prediction aggressiveness — materially changes the cost-benefit of formal vs. dynamic and the complexity of pipeline flush / replay verification. The methodology composition must accommodate both extremes.\n- **Power architecture (OPEN QUESTION 7).** Whether the core supports clock gating, power gating, retention, and/or DVFS is OPEN. The verification methodology must accommodate each combination. Assuming any specific power feature is invalid without evidence. Power intent (UPF/CPF) integration is conditional on this.\n- **Target implementation class (OPEN QUESTION 12).** Whether XH-1 targets ASIC tape-out or FPGA-only deployment fundamentally changes the verification methodology (DFT/ATPG scope, post-silicon layer, sign-off criteria). Treated here as an assumption (ASIC) pending resolution.\n\n## 128-Core Scalability\n\nReplication does not change which verification techniques apply to a single core, but it changes their cost-benefit and their composition.\n\n- **Core testbench scalability.** A core testbench written for a single instance must be instantiable inside a multicore wrapper without modification. This implies the testbench interfaces to the core through a standard transaction-level interface (or at least a clearly defined boundary), not through global signals.\n- **Cost-benefit of formal at replication.** Per-block formal verification has high fixed cost but, under bit-identical replication, a per-instance marginal cost of zero for the proven properties: once a control property is proven, it holds for every replicated instance of the same RTL. If the source is parameterized and elaborated with per-instance parameters, the proof must be re-established for any parameter set that differs. Under the replication-homogeneity assumption, formal's per-die marginal cost decreases as the replication count increases. The opposite is true for dynamic simulation at full coverage: a per-instance simulation cost must be paid for each replicated instance if all instances are to be exercised at full coverage, though the wall-clock growth is not strictly linear (the simulator elaborates and runs them in the same run, and the per-instance cost is amortized across the simulation kernel). The trade-off is a primary argument for the hybrid approach at 128-core scale, and the practical mitigation is sampling, not full per-instance coverage.\n- **Workload scaling.** Running the full core testbench in a 128-core environment is generally unnecessary; sampling (verify a subset of cores under full stimulus, sweep register state across all 128) is a more efficient pattern. The sampling strategy is a multicore concern but must be supported by the core testbench's reset/configuration interface. The statistical argument for the chosen sampling fraction (defect-escape probability vs. simulation cost) is a multicore-layer deliverable; the core testbench must support the interface.\n- **Coverage aggregation.** Per-core coverage data must be aggregated across the 128 instances. The core testbench should emit coverage in a form that is identifiable by instance index.\n- **Debug bandwidth.** 128 cores each generating waveform/log data is impractical. The core testbench must distinguish between detailed-trace mode (one core, full visibility) and lightweight-trace mode (128 cores, summary only).\n- **Compile and elaboration cost.** Elaboration of a 128-core testbench can dominate runtime. The core testbench should be elaborable standalone and in a multicore wrapper without recompiling the core RTL. Elaboration-time behavior (reset, clock tree, memory subsystem) can differ between 1-core and 128-core contexts; the core testbench must be valid in both.\n- **Testbench memory footprint.** 128 instances of a complex scoreboard and 128 reference models can dominate host memory. Memory-aware testbench design (shared reference models, deferred-check queues, summary-only trace mode) is required. This is acknowledged here and detailed in the multicore layer.\n- **Reference-model parallelism.** Spike and Sail are single-threaded by default. Running 128 reference models in lockstep with 128 RTL instances is a significant engineering problem. The recommended pattern is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling. This is a multicore-layer design constraint that the core testbench interface must support.\n- **Symmetry assumption revisit.** The replication-homogeneity assumption is load-bearing throughout this section. Heterogeneous cores (e.g., big/little, mixed application/accelerator, per-core ISA-subset configuration) invalidate the per-die marginal-cost argument and require per-variant verification, including per-variant testbenches, per-variant coverage, and per-variant formal. The methodology should be designed so that the additional cost of a variant is linear in the number of variants, not linear in 128 × variants.\n\n## Performance Considerations\n\nThe following are framework-level considerations. INSUFFICIENT EVIDENCE exists to provide numerical throughput estimates without XH-1-specific RTL data, simulator choice, and toolchain selection; any specific number is unsourced and is therefore not asserted here.\n\n- **Testbench throughput.** Dynamic simulation throughput determines how many random seeds can be run per unit wall-clock. Constrained-random at the core level is sensitive to simulator choice, RTL abstraction level, and coverage instrumentation; representative numbers are available in vendor documentation (commercial simulators) and in the open-source Verilator and cocotb benchmark literature, but are workload-dependent and cannot be quoted as typical without measurement on the XH-1 DUT.\n- **Compiler/optimizer.** Coverage instrumentation slows simulation. Production coverage runs and fast-turnaround debug runs should be separable.\n- **Co-simulation overhead.** Spike co-simulation overhead per committed instruction is non-trivial. Synchronization strategy (per-instruction vs. batched) is a tradeoff between debug granularity and throughput. At 128 instances, the cost of a per-instruction synchronization across all instances is prohibitive; batched or sampled synchronization is the practical pattern.\n- **FPGA prototype throughput.** FPGA prototype throughput is set by Fmax, which is determined by LUT count, routing, and the memory subsystem. Quoted MHz figures for an XH-1 core on a specific FPGA platform are not asserted here.\n- **Regression wall-clock budget.** A nightly regression that runs N seeds on a single core must fit within the available compute budget. Cloud burst capacity is typical.\n\nINSUFFICIENT EVIDENCE to provide numerical throughput estimates without XH-1-specific RTL data and toolchain selection.\n\n## Area Considerations\n\n- **On-die verification infrastructure.** XH-1 is described as a 128-core design. The verification *methodology* does not directly consume die area, but decisions made during verification (e.g., embedding assertions in RTL, adding observability muxes for trace, adding per-core DFT structures) do.\n- **Trace and debug infrastructure.** Core-level verification informs the design of trace, performance counters, and debug-transport blocks. Larger trace infrastructure costs area; the verification program must justify the cost in terms of debug productivity and post-silicon coverage.\n- **DFT (Design-for-Test) compatibility.** Scan chains, BIST, and MBIST are typically owned by a separate DFT program. Core verification should ensure that the core's DFT structures do not alter functional behavior in mission mode. If XH-1 is FPGA-only (OPEN QUESTION 12), DFT is out of scope.\n- **Replication-driven area optimization.** If the core's area is a major constraint (which is likely at 128 cores on a single die), the verification plan must protect against defects that would force a re-spin with re-spinned core area.\n\nINSUFFICIENT EVIDENCE to quantify area without access to XH-1's core area budget or process node.\n\n## Power and Energy Considerations\n\nThe following considerations apply **if and only if** XH-1 implements the corresponding power features. Each is OPEN QUESTION 7-conditional.\n\n- **Power-aware verification (conditional).** If the core supports power gating or retention, the core testbench must be able to drive representative switching activity to enable power estimation, and the testbench must verify that power-gated domains behave correctly (controlled wake, retention restore, ordering with respect to coherence traffic). UPF/CPF power intent integration with the simulator is the standard mechanism.\n- **Energy proportionality (conditional).** If the core supports clock gating at the instruction or block level, the testbench should verify that idle and active transitions behave correctly at the architectural interface (no observable state loss in idle, correct wake on interrupt).\n- **Thermal and DVFS interactions (conditional).** If XH-1 supports DVFS at the core level, voltage/frequency transitions and their effects on memory coherence, pipeline flush behavior, and CSR access must be verified. This is a cross-cutting concern between core and system verification.\n- **Power estimation accuracy.** Pre-silicon power estimation depends on workload realism. A core testbench that only runs synthetic random code will produce misleading power numbers.\n\nNo power feature is assumed. Verification provisions for any power feature must be activated only when OPEN QUESTION 7 is resolved.\n\n## Implementation Considerations\n\n- **Methodology composition.** The project must select one of the alternative strategies (or a hybrid) and codify it in the verification plan.\n- **Toolchain selection.** Open-source (Verilator, cocotb, SymbiYosys, `riscv-formal`, `riscv-dv`, Spike, RISCV-ISAC, GTKWave) vs. commercial (Synopsys VCS, Cadence Xcelium, Siemens Questa, JasperGold, VC Formal, FormalPro, Formality, Conformal, Palladium, Veloce, ZeBu) is a strategic decision. The choice cascades through licensing, hiring, training, and CI. These categories are not equivalent in role: Verilator and cocotb are simulation/cosim frameworks; SymbiYosys is a formal-verification flow; `riscv-formal` is an ISA-level formal property set; `riscv-dv` is a stimulus generator; RISCV-ISAC is a coverage tool; commercial emulators are hardware-assisted verification platforms distinct from FPGA prototyping. The choice within each role is independent.\n- **Reference model choice.** Spike (C++, instruction-accurate) vs. Sail (first-order imperative ISA-specification language with a RISC-V model) vs. a custom golden model. Spike is the most common reference; the project must document any corner-case behaviors of the chosen reference against the RISC-V specification.\n- **Coverage database.** Coverage data are stored in vendor-specific formats (e.g., Synopsys UCDB / DDC, Cadence IMC, Siemens UCDB) and may be aggregated into a custom backend (e.g., SQLite) for CI. The format is a tool-vendor concern; the aggregation backend is a CI concern. These are distinct choices.\n- **Test categorization.** Tests must be tagged by domain (base integer, M-ext, A-ext, F/D-ext, privileged, vector, …), by intent (sanity, corner, regression, performance), and by status (passing, known-fail, expected-fail).\n- **Seed management.** Random seeds should be reproducible. Seed pools should be partitioned into \"core regressions\", \"block regressions\", and \"fuzzing\".\n- **Bring-up sequencing.** Block-level → core-level → multicore → system. The core testbench is the transition point between block-level and system-level work.\n\n## Sign-Off Criteria\n\nA core verification sign-off is the point at which the methodology exits the coverage-closure funnel. The following are candidate criteria; numeric thresholds are starting examples to be refined in the verification plan and are not validated targets. The relationship to DRAFT status is explicit: the numbers below are illustrative starting points, not sign-off-ready thresholds.\n\n- **Functional coverage.** All defined cover points hit, with cross-coverage exercised. Threshold starting example: ≥ 95% cover-point hit, ≥ 90% cross-coverage hit; remaining items must be justified as unreachable or formally proven. Final threshold to be set in the verification plan.\n- **Code coverage.** Statement, branch, expression, toggle, and FSM coverage reported; excluded regions justified. Threshold starting example: ≥ 95% statement, ≥ 90% branch, with manual exclusion of reset/DFT/clock-gating logic. Final threshold to be set in the verification plan.\n- **Assertion coverage.** All SVA / immediate assertions vacuously satisfied or actively proven; assertion cover points hit. Open assertions must be reviewed and dispositioned.\n- **Formal closure.** Per-block formal sign-off, with proof closure recorded for each property. Inconclusive results must be either strengthened (via abstraction / decomposition) or accepted with risk justification; \"inconclusive\" is not equivalent to \"proven\".\n- **ISA compliance evidence.** All applicable tests in `riscv-arch-test` (the architectural test framework) and `riscv-tests` (the legacy compliance suite) pass. Signed certificate or equivalent record retained.\n- **Mutation testing.** Periodic injection of known-defect mutations into a frozen RTL revision; mutation-kill rate as a starting example: ≥ 80% of injected mutations detected by the testbench + assertions + formal. This number is a starting point and requires empirical validation on the XH-1 DUT. Equivalent-mutant handling (mutations that produce behaviorally identical RTL) must be defined; kill-rate targets are noisy and must be interpreted with caution. Final threshold and operator set to be set in the verification plan.\n- **Reference model validation.** Spike or Sail validated against the chosen RISC-V compliance tests before being used as a checker; validation evidence archived.\n- **Testbench self-check verification.** Every scoreboard check has at least one negative test that is expected to fail.\n- **Reproducibility.** Every sign-off run is reproducible from committed source, seed, and tool version.\n- **CDC/RDC closure.** CDC and RDC analysis clean for all core-level crossing structures.\n- **Lint clean.** Lint clean at the policy-defined severity level.\n- **Reset verification.** All reset domains and reset sequences verified; reset recovery verified.\n\n## Verification Considerations\n\nCore verification is itself subject to verification — the verification of the verification environment:\n\n- **Testbench self-checking.** All scoreboard checks must be end-of-test verifiable; every check should be paired with at least one negative test that is expected to fail.\n- **Reference model validation.** The reference model (Spike, Sail) must be validated against the chosen RISC-V compliance tests before being used as a checker.\n- **Coverage sanity.** Cover points must be reviewed to ensure they are not trivially self-firing.\n- **Mutation testing.** A subset of RTL defects should be intentionally injected to confirm the testbench catches them. See `research/10-verification/fault-injection.md`. Equivalent-mutant handling must be defined.\n- **Coverage baseline stability.** Coverage baselines drift as RTL changes; a baseline-management policy (when a baseline is re-established, by whom, with what approval) is required.\n- **Tool qualification.** If the verification tools are used in any safety-relevant context, tool qualification (per ISO 26262, IEC 61508, DO-254, or similar) is required. The applicability of these standards to XH-1 is OPEN QUESTION 8. Tool qualification is a verification-of-verification concern, not only a sign-off checkbox.\n- **Version control and baselines.** Every test run must be reproducible from committed source. Coverage baselines must be versioned.\n- **Testbench verification at 128-instance scale.** The testbench itself has bugs. Verification of the testbench (negative tests, mutation of the testbench, equivalence of testbench revisions) is part of the methodology. At 128 instances, the testbench's own correctness becomes a non-trivial verification problem and is acknowledged here as a multicore-layer concern that the core testbench interface must support.\n\n## Software Considerations\n\n- **Toolchain integration.** The core verification environment must integrate with the software toolchain (compiler, assembler, linker, Spike, debug server) to enable co-simulation and to run real programs.\n- **Firmware.** Firmware that runs on the core (boot code, exception handlers, interrupt dispatchers) is itself verification target; the core testbench should support loading and running firmware images.\n- **Operating system implications.** A core that supports an OS (Linux, RTOS) must be verified with syscall-heavy workloads. The core testbench should provide OS-aware run modes.\n- **Debug and trace protocols.** The core's debug-transport module (per the RISC-V Debug specification) must be exercised. JTAG or cJTAG interactions with the core's debug module are typically a core-level concern.\n- **Performance analysis tools.** Tools that analyze trace data, performance counters, and profile information should be able to consume the artifacts the core testbench produces.\n- **Software developer ergonomics.** The verification team's interfaces (test writers, coverage reviewers) should be ergonomic. A \"developer can run a single test and see results in under a minute\" goal is a useful operational target.\n\n## Security Considerations (Out of Scope, but Flagged)\n\n**ASSUMPTION (out-of-scope for this revision):** Transient-execution / side-channel verification (e.g., Spectre-class, Meltdown-class, and constant-time verification) is treated as out of scope for the XH-1 core verification layer for this revision. For a 128-core design with privilege boundaries, this is a real and high-risk concern. This is flagged as a high-risk gap and as a candidate for a follow-up document. The current methodology is sufficient for functional and microarchitectural correctness but does not, by itself, provide security assurance.\n\n## Proposed Direction\n\nPROPOSAL — A **hybrid methodology** for XH-1 core verification, with the following structure:\n\n1. **Dynamic simulation** as the primary verification engine, with the framework (UVM vs. lightweight) gated on OPEN QUESTION 3 (tool posture), OPEN QUESTION 4 (team size and timeline), and the design-language choice (OPEN QUESTION 11).\n - If UVM: pair with `riscv-dv` as the instruction-stream generator and Spike or Sail as the reference model.\n - If lightweight: pair `riscv-dv` (or a custom generator) with cocotb / Verilator and Spike.\n2. **Formal verification** applied to control-dominated blocks: fetch, decode, CSR file, MMU/TLB, exception/interrupt delivery, and the debug module. Coordinate with `research/10-verification/formal-verification.md` for tool selection. Apply `riscv-formal` as an ISA-level formal layer where the open-source flow is in scope. Sequential equivalence checking between major RTL revisions is a complementary activity.\n3. **Co-simulation against a RISC-V reference model** (Spike or Sail) for ISA-compliance evidence. Coordinate with `research/10-verification/isa-compliance.md`. Pin the RISC-V ISA, Privileged ISA, and Debug spec versions. At 128 instances, use a shared reference model with a deferred-check queue or sampled synchronization.\n4. **Directed tests** for XH-1 microarchitectural corner cases that random cannot reach economically. The corner cases to target are conditional on the in-order / out-of-order answer in OPEN QUESTION 2.\n5. **FPGA prototyping** for throughput-driven verification and as a substrate for fault injection and multicore bring-up. Commercial emulation is a conditional alternative based on licensing posture (OPEN QUESTION 3) and workload.\n6. **Regression infrastructure** with version-controlled seeds, baselines, and coverage data. Coordinate with `research/10-verification/regression.md`.\n7. **Coverage closure** as an explicit, tracked funnel with the sign-off criteria in the Sign-Off Criteria section, including baseline-management policy. Numeric thresholds are deferred to the verification plan and the numbers in the Sign-Off section are starting examples, not validated targets.\n8. **Reusable testbench architecture** so the core testbench composes inside the multicore testbench without modification, supporting both the per-die cost argument and the sampling strategy. The interface must support shared reference models, summary-only trace mode, and per-instance coverage identification.\n9. **Mutation testing** as a periodic check that the testbench is still doing useful work, with explicit equivalent-mutant handling. Coordinate with `research/10-verification/fault-injection.md`.\n10. **Standard core-verification adjuncts:** lint, CDC/RDC analysis, X-propagation handling, reset verification, and (conditional on OPEN QUESTION 7) UPF/CPF power-intent integration.\n\nThe hybrid is proposed; the specific framework within the hybrid (UVM vs. lightweight) is gated on OPEN QUESTION 3, OPEN QUESTION 4, and OPEN QUESTION 11, and the power-feature sub-portion is gated on OPEN QUESTION 7. The implementation-class sub-portion (DFT/ATPG scope, post-silicon layer) is gated on OPEN QUESTION 12. The document does not pick between alternatives that the OPEN QUESTIONS do not yet support.\n\n## Confidence\n\n- **High confidence:** The general methodology composition (dynamic + formal + co-simulation + directed + prototype + regression + coverage closure, with the standard adjuncts: lint, CDC/RDC, X-propagation, reset) corresponds to the IEEE 1800.2 UVM standard, the IEEE 1800 SystemVerilog assertion subset, and the RISC-V International published ISA / Privileged / Debug specifications, and is broadly accepted in industrial practice. The references supporting each component are cited at the point of use.\n- **High confidence (conditional):** The 128-core scaling argument (formal has near-zero per-instance marginal cost under bit-identical replication homogeneity; dynamic at full coverage has per-instance cost) is a direct consequence of the replication-homogeneity assumption and does not require empirical data. The parameter-set caveat (re-elaboration for differing parameters) is acknowledged.\n- **Medium confidence:** Specific tool selections, sign-off numeric thresholds, and area/power tradeoffs, given the absence of repository evidence on the OPEN QUESTIONS.\n- **Low confidence (OPEN):** XH-1's tool licensing posture, core microarchitecture details, ISA profile selection, team size, design language, power architecture, and target implementation class. These drive the specifics within the proposed hybrid.\n\n## Open Questions\n\n1. What RISC-V profile does the XH-1 core implement? Candidate RISC-V profiles are defined against a base XLEN: e.g., RVA23U64 / RVA23S64 are 64-bit application / supervisor profiles, while a 32-bit core would follow a different profile track (e.g., RVA22U32). The reference model (Spike variant, Sail configuration) and the compliance test scope follow from this choice.\n2. Is the XH-1 core in-order or out-of-order? The verification plan complexity differs substantially. The instructions explicitly disallow assuming either without repository evidence.\n3. What is XH-1's toolchain posture (open-source only, commercial EDA, mixed)?\n4. What is the team size and timeline? Methodology cost scales with team capacity.\n5. What is the core's microarchitectural contract — what blocks are in scope for unit verification, and what is the boundary between core and multicore? See `research/10-verification/unit-verification.md` and `research/10-verification/multicore-verification.md`.\n6. Are the 128 cores architecturally identical, or are there variants (big/little, mixed application/accelerator, per-core ISA-subset configuration)? Heterogeneity changes the testbench architecture and invalidates the per-die cost-leverage argument.\n7. What is the core's power architecture — is clock gating, power gating, retention, or DVFS at the core level, and what are the state-retention semantics?\n8. What is XH-1's relationship to functional safety standards (ISO 26262, IEC 61508, DO-254, or none)?\n9. How will the XH-1 core's trace, debug, and performance-counter interfaces be designed, and how will core verification consume them?\n10. What is the expected defect rate target at tape-out, and how does the verification program demonstrate closure against that target?\n11. What is the XH-1 design language (SystemVerilog, Verilog-2001, Chisel/Spinal HDL, Bluespec), and does verification target the HDL source level, the generated-RTL level, or both?\n12. What is XH-1's target implementation class — ASIC tape-out, FPGA-only deployment, or both? This determines DFT/ATPG scope, post-silicon validation depth, and sign-off criteria.\n13. What physical implementation tool flow (synthesis, PnR, timing) is envisioned, and what are the timing-closure constraints (target clock frequency, slack budget)?\n\n## Sources\n\n- IEEE 1800.2-2020, *Standard for Universal Verification Methodology Based on Accellera SystemVerilog 3.1*. Defines UVM.\n- IEEE 1800-2017, *Standard for SystemVerilog — Unified Hardware Design, Specification, and Verification Language*. Defines SVA and immediate assertions.\n- RISC-V International, *The RISC-V Instruction Set Manual, Volume I: Unprivileged Architecture* (versioned).\n- RISC-V International, *The RISC-V Instruction Set Manual, Volume II: Privileged Architecture* (versioned).\n- RISC-V International, *RISC-V External Debug Support* (versioned).\n- RISC-V International, *RISC-V Profiles* (versioned, e.g., RVA23).\n- Symbiotic EDA / YosysHQ, *SymbiYosys* documentation (open-source formal verification flow).\n- `riscv-formal` framework (Verilog, BSD-licensed), maintained at github.com/YosysHQ/riscv-formal.\n- `riscv-dv` instruction-stream generator, maintained at github.com/google/riscv-dv.\n- `riscv-tests` and `riscv-arch-test`, maintained by RISC-V International.\n- Spike (the official RISC-V ISA simulator), maintained at github.com/riscv-software-src/riscv-isa-sim.\n- Sail RISC-V formal specification, maintained at github.com/rems-project/sail. Sail is a first-order imperative ISA-specification language; the RISC-V model is generated from the RISC-V ISA formal definition.\n- Kami project (Cambridge), formal RISC-V ISA specification in Coq with a Bluespec front-end. Project publications and documentation should be cited from the project's own canonical sources.\n\nQuantitative throughput, area, power, and timing claims are not asserted in the body. The XH-1 repository documents referenced (under `research/10-verification/`) are placeholders (\"SOON\") and do not provide factual content to cite. No measurements, benchmarks, processor capabilities, URLs outside the repositories named above, or experimental results have been invented.","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":11757,"completion_tokens":11041,"total_tokens":22798,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":135,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round4.md b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round4.md new file mode 100644 index 0000000..2035a2b --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round4.md @@ -0,0 +1,385 @@ +# Core Verification + +## Status + +DRAFT — Research framework established; all claims are explicitly classified by epistemic status (FACT with citation, ASSUMPTION, PROPOSAL, RECOMMENDATION, OPEN QUESTION, or INSUFFICIENT EVIDENCE). Concrete sign-off criteria, tool selection, and quantitative thresholds are deferred until the OPEN QUESTION inputs are resolved. The numeric thresholds appearing in the Sign-Off Criteria section are illustrative starting points for the verification plan and are not validated targets. Every threshold below is restated with this caveat at the point of use so that the sign-off list is internally consistent with this status. + +## Abstract + +This document investigates verification strategies for the XH-1 core, defined here as a single processing core instance within a custom 128-core RISC-V processor. Core verification is the layer of the verification methodology stack concerned with demonstrating that one core conforms to its architectural specification, microarchitectural contract, and integration assumptions prior to and independent of multi-core interactions. For a 128-core design, the core is the unit of replication; defects undetected at this layer replicate 128-fold across the die, amplifying area, power, yield, and debug cost. This document frames the research questions, surveys existing approaches (simulation, formal verification, hardware-assisted verification, FPGA prototyping, emulation), analyzes their applicability to a custom RISC-V core, and identifies the gaps that the XH-1 project must close given that the existing related documents in `research/10-verification/` are placeholders ("SOON"). + +## Research Question + +What verification methodology, toolchain, environment abstraction, coverage model, and sign-off criteria are appropriate for verifying a single XH-1 core such that the methodology: + +1. Provides evidence that the core implements the RISC-V ISA subset selected for XH-1 (see `research/10-verification/isa-compliance.md`). +2. Provides evidence that the core's microarchitecture meets timing, area, and power targets for a design that will be replicated 128 times. +3. Scales economically — verification effort, compute cost, and human review time must not grow linearly with the 128-core replication factor. +4. Produces artifacts (testbenches, coverage models, formal proofs, regression suites) that compose cleanly with the surrounding verification layers (unit, multicore, system, post-silicon). +5. Is implementable given that the related XH-1 documents in area 10 are not yet populated, indicating a greenfield verification program. + +## Background + +Verification of a processor core is conventionally decomposed into: + +- **Unit verification** — verification of individual blocks (ALU, FPU, register file, branch predictor, fetch/decode structures, load-store unit, replay logic, CSR file, MMU/TLB, caches). See `research/10-verification/unit-verification.md`. +- **Core verification** — verification of the integrated core as a single design under test (DUT), exercising instruction-stream and exception behavior at the architectural interface, plus internal microarchitectural invariants. +- **Multicore verification** — coherence, consistency, interconnect, and shared-resource correctness across multiple core instances. See `research/10-verification/multicore-verification.md`. + +For XH-1, the core is the replication unit. The cost model must distinguish two quantities that this document often refers to together as "cost": + +- **Defect-escape cost (multiplied by replication).** The downstream cost of an undetected defect (silicon debug, re-spin, field failure, yield loss). Under the replication-homogeneity assumption, an undetected core defect is replicated 128-fold in silicon, multiplying its downstream impact. +- **Verification effort cost (paid once).** The engineering cost of running the testbench, discharging formal properties, maintaining the regression, and writing tests. This is paid against one RTL; the fix, once made, is replicated. + +The phrase "every bug caught at the core layer is caught 128 times" is shorthand for the first quantity: every defect that escapes core verification multiplies its downstream cost by the replication count. It is not a claim that the verification effort itself runs 128 times — the testbench runs against the RTL once. + +**ASSUMPTION (replication homogeneity):** The 128 XH-1 cores are architecturally identical instances of a single RTL core produced by elaboration of the same source. Heterogeneous cores, per-core configuration, or runtime variants invalidate the replication-leverage argument on defect-escape cost and require per-variant verification. This is treated as an assumption because no repository evidence contradicts it; OPEN QUESTION 6 below revisits it. The "per-instance marginal cost of zero" claim for proven formal properties holds for bit-identical replicated RTL and refers to the marginal cost of *re-running* a proof against another elaborated instance of the same source — not to the engineering cost of constructing the proof. If the source is parameterized and elaborated with per-instance parameters that differ, the proof must be re-established for any parameter set that differs, and that re-establishment cost is not zero. + +**ASSUMPTION (target implementation class):** XH-1 targets ASIC tape-out. FPGA-only deployment would change the methodology (no DFT/ATPG, different sign-off criteria, no post-silicon bring-up in the same sense). This assumption is conditional and is revisited in OPEN QUESTION 12. + +**ASSUMPTION (verification language availability):** The verification language of the testbench is the same as or interoperable with the design language. This is revisited in OPEN QUESTION 11. + +Core verification draws on several formal and semi-formal techniques: + +- **Dynamic simulation** with constrained-random or directed tests, against an ISA reference model, with functional coverage. +- **Static (formal) verification** — model checking (bounded model checking, k-induction, IC3/PDR), equivalence checking (combinational and sequential), and property proving against the RTL. +- **Hardware-assisted verification** — FPGA prototyping and emulation (commercial emulators, open-source flows such as FireSim on Amazon EC2 F1), trading compile time for runtime speed. +- **Co-simulation** — two simulation engines running in lockstep (e.g., an RTL simulator and an architectural reference-model engine such as Spike) synchronized at defined events. This is distinct from **reference-model checking**, in which the reference model is integrated as a transactor within a single simulator; the latter is a common industrial pattern and is more accurately described as reference-model checking, not co-simulation in the strict sense. +- **Waveform and assertion-based methodologies** — SystemVerilog Assertions (SVA), PSL, custom monitors, end-of-test checkers. + +Coverage is a central concern. Functional coverage, code coverage (line, branch, expression, toggle, FSM), and cross-coverage of architectural state are the standard dimensions. For a custom ISA implementation, the relationship between ISA compliance coverage and microarchitectural coverage must be defined explicitly. + +Standard core-verification concerns that must be addressed in addition to instruction-stream behavior: + +- **Lint** — static analysis of RTL style, common bug patterns, and synthesizability (e.g., Verilator lint, commercial lint tools). A prerequisite to dynamic and formal work. +- **Clock-domain crossing (CDC) and reset-domain crossing (RDC)** — analysis of signals crossing asynchronous clock or reset domains. Typically owned by a separate methodology but the core must be CDC/RDC-clean for integration. +- **X-propagation** — controlled handling of unknown ('X) propagation in simulation, using X-pessimism removal or X-init optimization, to avoid false coverage from masked X's. +- **Reset verification** — power-on reset, software reset, debug reset, partial reset, and reset-domain interaction. +- **Power intent** — if power features are present (see OPEN QUESTION 7), UPF/CPF power intent formats integrate with simulation to model power-gated, retention, and isolation behavior. +- **DFT/ATPG interaction** — scan, BIST, and MBIST structures must not alter mission-mode behavior; their interaction with the core testbench is a coordination point. + +These are not optional; their absence in a verification plan is a known gap in pre-silicon methodology. + +## Existing Approaches + +The following approaches are established in industry and academic practice. Their applicability to XH-1 is analyzed; none is recommended in isolation. + +### A. Constrained-Random Simulation with UVM + +**FACT:** Universal Verification Methodology (UVM) is standardized as IEEE 1800.2-2017, *Standard for Universal Verification Methodology* [IEEE 1800.2-2017], and is widely used in industrial block- and core-level processor verification for constrained-random stimulus generation, reference model integration, self-checking scoreboards, functional coverage, and phasing. The Accellera UVM reference implementation, which is the source code that vendors and users actually consume, is distributed under the Apache License 2.0; the IEEE 1800.2-2017 standard document itself is an IEEE publication and is not freely redistributable. The two are distinct artifacts and this distinction is load-bearing when characterizing "open-source viability" below. + +**ASSUMPTION (conditional on design language):** UVM is applicable only if the XH-1 design language supports SystemVerilog class constructs — either the source is SystemVerilog, or the verification environment targets SystemVerilog regardless of source language. If the design language is Verilog-2001 only, or if the project commits to a Chisel/Spinal HDL or Bluespec-only flow, UVM is not directly applicable at the HDL-source level and an alternative testbench harness is required. This assumption is revisited in OPEN QUESTION 11. + +**PROPOSAL (applicability to XH-1):** Apply UVM as the primary dynamic verification environment for XH-1 core-level testing, paired with a chosen ISA reference (Spike or Sail). Sequence generation over the RISC-V ISA state space; reference model in co-simulation; scoreboard for architectural state comparison per committed instruction; explicit cover points for instruction × operand-mode × privilege × exception interactions. + +### B. Directed and Hand-Written Tests + +**FACT:** Constrained-random alone does not reach corner-case coverage of complex pipelines, corner-case CSR interactions, or rare exception sequences. Industry practice supplements with directed tests written by architects and designers. + +**PROPOSAL (applicability to XH-1):** A XH-1-specific directed test suite for pipeline hazard combinations specific to the XH-1 microarchitecture, corner cases in the chosen RVA profile subset, reset, wake, low-power entry/exit, and debug transport (if present at core level). + +### C. Formal Property Verification + +**FACT:** Bounded model checking, k-induction, and IC3/PDR are applied to control-dominated blocks: fetch, decode, CSR, MMU/TLB, branch predictor, retirement, and exception delivery. Commercial model checkers include Cadence JasperGold, Synopsys VC Formal, and Siemens FormalPro. Open-source flows include SymbiYosys (SBY), based on the Yosys synthesis and formal toolchain. Equivalence checking is a distinct formal category: tools include Synopsys Formality and Cadence Conformal, used to verify that two RTL representations (e.g., pre- and post-synthesis, or pre- and post-retiming) are functionally equivalent. Equivalence checkers are not model checkers and are not used for property proving. + +**PROPOSAL (applicability to XH-1):** Per-block formal sign-off for the control-dominated blocks listed above, end-to-end pipeline flushing, exception priority, and CSR read/write semantics. Formal proofs of privileged ISA semantics (e.g., `mstatus`, `sstatus`, `satp`, `mip`/`sip` behavior) provide a high-confidence anchor that random tests cannot match in isolation. Sequential equivalence checking between major RTL revisions is a separate, complementary activity. + +**ASSUMPTION (RISC-V privileged-ISA versioning):** The CSR semantics used as formal-proof targets correspond to the specific RISC-V Privileged ISA version adopted by XH-1. The unprivileged ISA, privileged ISA, and Debug specification each have versioned releases tracked by RISC-V International; the verification plan must be pinned to those versions. The specific version is OPEN until resolved (see OPEN QUESTION 1 and the Sources section). + +**PROPOSAL (applicability to XH-1):** Coordinate with `research/10-verification/formal-verification.md` for tool selection (commercial model checkers vs. SymbiYosys / SBY) and with `research/10-verification/isa-compliance.md` for the privileged-ISA version. + +### D. Reference-Model Checking and Co-Simulation + +**FACT:** Comparing DUT architectural state against Spike (the official RISC-V ISA simulator, C++ implementation) or Sail (a first-order imperative language designed for ISA specification, with a RISC-V model generated from the RISC-V ISA formal definition) per instruction is a widely deployed pattern. The terminology matters: + +- **Co-simulation (strict sense):** two independent simulation engines (e.g., an RTL simulator and Spike) running in lockstep, synchronized at defined events. The DUT may run ahead of the reference using a deferred-check queue, with results compared at synchronization points. +- **Reference-model checking:** a single simulator that incorporates the reference model as a transactor or foreign-language call. This is a common industrial pattern but is not co-simulation in the strict sense. + +**PROPOSAL (applicability to XH-1):** Required for ISA-compliance evidence. See `research/10-verification/isa-compliance.md`. The specific architecture proposed is **reference-model checking within a single simulator** (Spike invoked as a transactor from the UVM or lightweight harness) as the default for day-to-day runs, with **co-simulation in the strict sense** retained as a cross-check on a sampled subset of tests to detect synchronization artifacts that a single-simulator harness could mask. Reference model parallelism is a 128-core scaling concern: Spike and Sail are single-threaded by default; running 128 reference models in lockstep with 128 RTL instances is itself a significant engineering problem. The pattern of choice is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling (multicore-layer decision) with the core testbench supporting the interface. + +### E. FPGA Prototyping and Emulation + +**FACT:** FPGA-based prototyping runs the RTL at higher throughput than software simulation, and emulation platforms (commercial: Palladium, Veloce, ZeBu; open-source: FireSim on Amazon EC2 F1) provide a middle ground. The throughput multiplier relative to software simulation is workload- and instrumentation-dependent. + +**PROPOSAL (applicability to XH-1):** Use FPGA prototyping to speed up core testbench bring-up by running entire test programs end-to-end, to enable fault-injection campaigns that are infeasible in simulation (see `research/10-verification/fault-injection.md`), and as the natural place to begin validating the multicore layer; core verification is the upstream prerequisite. Commercial emulation is a viable alternative for very large workloads or when observability requirements are relaxed; the decision depends on licensing posture (OPEN QUESTION 3) and workload characteristics. No specific throughput multiplier for the XH-1 DUT is asserted here; INSUFFICIENT EVIDENCE without XH-1-specific RTL and platform data. + +### F. Regression Infrastructure + +**FACT:** Regression management — version control of testbenches, seeds, baselines, coverage databases — is a load-bearing component of any non-trivial core verification program. See `research/10-verification/regression.md`. + +**PROPOSAL (applicability to XH-1):** The XH-1 regression infrastructure must support a single-core nightly run, a multicore run, and a system run with shared seeds where appropriate. + +### G. Coverage Closure Methodology + +**FACT:** Coverage closure combines functional coverage (defined by the verification plan), code coverage (uncovered RTL), and assertion coverage. Industry practice treats coverage closure as a managed funnel: cover points → cross coverage → corner case identification → directed/random test addition → formal where possible. Coverage baselines drift as RTL changes; a baseline-management policy is required. + +**PROPOSAL (applicability to XH-1):** Codify the funnel explicitly in the verification plan with tracked stages, baseline-management policy, and exit criteria per stage. + +### H. Assertion-Based Verification + +**FACT:** SystemVerilog Assertions (IEEE 1800-2017, *Standard for SystemVerilog*) and immediate assertions encode design intent and serve both as live checkers and as inputs to formal tools. + +**PROPOSAL (applicability to XH-1):** Define an SVA library for XH-1 invariants (e.g., "no retired instruction has a privileged side effect inconsistent with its mode", "no speculative architectural state is observable at the architectural interface"). + +### I. Open-Source ISA-Level Formal Verification + +**FACT:** The `riscv-formal` framework (YosysHQ / Symbiotic EDA, Verilog, BSD-licensed) provides a checkable RISC-V ISA specification encoded as a set of formal properties that are conjoined with a wrapper around the DUT and discharged by an SMT solver. The Kami project (Cambridge) provides a formal RISC-V ISA specification in Coq with a Bluespec front-end. INSUFFICIENT EVIDENCE to assert at the FACT level that `riscv-formal` is "directly applicable" to a specific XH-1 microarchitecture and ISA profile without further evaluation: applicability depends on the DUT's memory and CSR interface matching the framework's wrapper assumptions, on the chosen RISC-V profile, and on the unprivileged/privileged ISA versions adopted. + +**PROPOSAL (applicability to XH-1):** Evaluate `riscv-formal` as a complement to block-level formal. It targets ISA-compliance evidence with open-source tooling and is particularly relevant given the greenfield status of the XH-1 verification program, subject to the applicability evaluation above. + +### J. Open-Source Constrained-Random Generator + +**FACT:** Google's `riscv-dv` is an open-source UVM-based constrained-random instruction-stream generator for RISC-V, widely used in industrial and academic flows. The license should be confirmed against the current repository state at the time of toolchain selection; this document does not assert a specific license version. + +**PROPOSAL (applicability to XH-1):** Evaluate `riscv-dv` as a generator either within a UVM environment or as a stimulus source for a lightweight harness. + +### K. RISC-V Architectural Test Framework + +**FACT:** RISC-V International maintains `riscv-arch-test` as the architectural test framework, which is the conformance test suite against which cores are signed off for profile compliance. `riscv-tests` is a related umbrella repository containing `riscv-arch-test` as well as additional tests; the relationship is not "legacy vs. replacement" — `riscv-tests` is the broader collection and `riscv-arch-test` is the architectural conformance subset. The two should be treated as distinct but related inputs, both of which produce evidence in the sign-off record. The framework's coverage analyzer, `RISCV-ISAC`, is an instruction-stream coverage analyzer used to confirm that the architectural test signatures cover the required instruction/operand/privilege combinations; it is not a generic functional-coverage database and is not a substitute for the core testbench's own coverage model. + +**PROPOSAL (applicability to XH-1):** Use `riscv-arch-test` for the ISA-compliance certificate, with `RISCV-ISAC` analysis of the architectural test signatures as part of the coverage evidence, and `riscv-tests` more broadly as a regression input. The license and current revision of each must be confirmed at toolchain selection. + +## Alternative Designs + +The following alternative *strategies* for the core verification methodology itself are candidates for XH-1. They are not mutually exclusive; the project must select a composition. + +### Alternative 1: UVM-Centric with Formal Supplements + +- **Approach:** UVM as the workhorse for dynamic verification, with formal applied to control-dominated blocks. +- **Strengths:** Industry-standard methodology, well-understood coverage and sign-off conventions, mature ecosystem; integrates with `riscv-dv` as a generator and `riscv-formal` as an ISA-level formal layer. +- **Weaknesses:** UVM has a steep learning curve and high setup cost. For a 128-core design, the UVM environment must be designed for reuse to avoid 128× rewrite cost. Requires SystemVerilog (OPEN QUESTION 11). Implies commercial simulator licensing in most deployments, because open-source SystemVerilog simulation support (Verilator) has limited UVM-class support relative to commercial offerings; the open-source viability of this alternative is therefore medium, not high. + +### Alternative 2: Lightweight Python/C++ Test Harness with Reference-Model Checking + +- **Approach:** A generator produces instruction streams and a C++ testbench (e.g., based on cocotb, Verilator, or a custom framework) compares architectural state against Spike. +- **Strengths:** Lower setup cost than UVM. Excellent generator productivity. Easy integration with regression infrastructure. Open-source-friendly. `riscv-dv` can serve as the generator. Verilator-based flows are not bound to SystemVerilog UVM at the source level. +- **Weaknesses:** Verilator's debug, SVA, and functional-coverage support are not equivalent to commercial simulators; cocotb+Verilator is therefore not a drop-in replacement for UVM, and the trade-off should be evaluated against the XH-1 debug and coverage requirements. Coverage closure tooling is less standardized than the UVM vendor ecosystem. The reference-model-checking pattern (Section D) is the natural fit for this harness. +- **Application to XH-1:** Reasonable choice if XH-1 has a small team, an open-source toolchain preference, or a constraint on commercial EDA licensing. + +### Alternative 3: Formal-First + +- **Approach:** Formal proofs form the primary verification evidence; dynamic simulation is used for reach and to validate properties that formal cannot discharge. +- **Strengths:** Highest assurance per proven property. Strong fit for control logic and CSR behavior. `riscv-formal` provides an open-source ISA-level formal layer subject to the applicability evaluation in Section I. Under the replication-homogeneity assumption, proven properties transfer to all 128 instances at near-zero marginal *re-proof* cost (the proof is re-run, not re-constructed, against another elaborated instance of the same RTL); the engineering cost of constructing the proof is paid once. +- **Weaknesses:** Pipeline formalization is hard. State-space complexity grows combinatorially for large cores. May not reach proof closure on timing-agnostic models without abstraction. Inconclusive results are not equivalent to failure; risk-acceptance must be tracked. + +### Alternative 4: Emulation-Centric + +- **Approach:** Skip large-scale software simulation; use FPGA/emulation throughput as the primary engine. +- **Strengths:** Excellent long-workload capability; native fault-injection support. +- **Weaknesses:** Poor waveform/observability; slow compile/turnaround; harder to drive corner-case stimulus; debug cycle is slow. + +### Alternative 5: Hybrid (PROPOSAL — See Proposed Direction) + +A combination of a dynamic framework (Alternative 1 or Alternative 2) for stimulus and reach, formal supplements (Alternative 3, including `riscv-formal` where applicable) for control logic, and FPGA prototyping/emulation for throughput and multicore bring-up. + +## Comparison + +The comparison is qualitative and is intended to make the trade-offs explicit, not to produce a numeric ranking. INSUFFICIENT EVIDENCE exists to rank absolute quality without XH-1-specific RTL data, tool licensing posture, and team size. The Hybrid column is a composition of the others and is not an independent point of comparison; the table is structured to show how each pure strategy performs on the dimensions, with the Hybrid column indicating how the strengths compose. + +| Dimension | UVM-Centric | Lightweight Harness | Formal-First | Emulation-Centric | Hybrid | +|---|---|---|---|---|---| +| Setup cost | High | Low | High | Medium | High | +| Per-test runtime | Slow | Slow–Medium | N/A (proof time) | Fast | Mixed | +| Corner-case reach | Strong (with seeds) | Strong (with fuzzing) | Strong (when discharged) | Weak–Medium | Strong | +| Assurance on control logic | Medium | Medium | High | Low–Medium | High | +| Debug ergonomics | Good (waveform, logs) | Good (caveat: Verilator/cocotb debug support not equivalent to commercial) | Strong counterexamples | Weak | Mixed | +| Open-source viability | Medium (UVM standard is IEEE-published; UVM reference implementation is Apache 2.0; simulators are usually commercial) | Strong | Medium | Medium | Medium | +| Fit for 128-core reuse | High (if designed so) | High (if designed so) | High (under replication homogeneity) | High | High | +| Pre-tape-out risk reduction | Strong | Medium–Strong | Strong (partial; inconclusive ≠ proven) | Medium | Strong | + +## Advantages + +The advantages of investing in a robust core verification methodology for XH-1: + +- **Replication leverage on defect-escape cost.** Under the replication-homogeneity assumption, an unfixed defect in the core RTL is replicated 128-fold in silicon, multiplying its downstream (silicon-debug, re-spin, field-failure, yield-loss) cost. Verification effort itself is paid once against one RTL; the defect, if found, is fixed once; the fixed RTL is then replicated. The benefit of core-level detection is the avoidance of multiplied post-silicon cost. +- **Cleaner multicore verification.** A well-verified core reduces the multicore verification problem to coherence, consistency, interconnect, and shared resources. See `research/10-verification/multicore-verification.md`. +- **Foundation for post-silicon.** First-stage silicon bring-up is dramatically simpler when the core is well-understood and the debug interfaces (assertions, traces, scan) are well-designed. Post-silicon validation feeds back into pre-silicon coverage targets; this feedback loop is not detailed in this document and is left to the post-silicon layer. +- **Reusable testbench assets.** Generator, scoreboard, coverage model, and SVA library can be reused across cores if XH-1 has core variants, or across revisions of the same core. +- **Formal evidence on critical control.** Privileged ISA semantics are hard to verify by random simulation; formal proofs on these blocks provide a high-confidence floor. + +## Disadvantages + +Costs and risks of a robust core verification program: + +- **Engineering cost.** A UVM-class environment plus formal plus emulation is a substantial program. The claim that it "can easily exceed the RTL design cost for a single core" is folkloric and is presented here as a planning consideration, not as a measured fact. +- **Tooling lock-in.** Heavy UVM usage often implies commercial simulator licensing; this may not match XH-1's tool posture. Formal tool lock-in is similar: commercial model checkers differ in property language and run scripts. +- **Slow turnaround.** Constrained-random simulation with full coverage at the core level is slow; coverage closure can take many cycles of "add test, rerun, close hole". Coverage baselines drift with RTL changes, requiring re-establishment. +- **Reference model dependence.** Spike, Sail, or any reference may diverge from the RISC-V specification in interpretation of ambiguous or profile-specific corners; the verification program must track the chosen RVA profile carefully and pin the spec version of the reference. +- **Coverage false confidence.** Hitting a cover point is not the same as proving correctness; the verification plan must distinguish cover and check. +- **128-core replication pressure.** If the core testbench is not designed to compose with the multicore testbench, redundant effort emerges. +- **Inconclusive formal ≠ proven.** Formal results of "inconclusive" must be risk-accepted, not treated as success. +- **Equivalent mutants.** Mutation-kill rates are affected by equivalent mutants (mutations that produce behaviorally identical RTL); kill-rate targets are therefore noisy and must be interpreted with caution. + +## XH-1 Considerations + +XH-1-specific factors that constrain the methodology choice: + +- **128-core replication.** The verification plan must be replication-aware: the *testbench* should be agnostic to whether one or 128 cores are instantiated. This implies strong separation of (a) the core DUT, (b) the core testbench, and (c) the multicore wrapper. +- **Custom design.** No third-party core is being verified; there is no vendor's pre-built verification IP. All environments must be built or assembled. +- **Repository context.** The related verification documents are all placeholders ("SOON"). This is a greenfield verification program; the XH-1 project has the opportunity — and the burden — to define the methodology from first principles. +- **RISC-V ISA.** The RISC-V ecosystem provides mature reference models (Spike, Sail), compliance / architectural test frameworks (`riscv-arch-test`, `riscv-tests`), and open-source verification tooling (`riscv-formal`, `riscv-dv`, `RISCV-ISAC`). These are the natural anchors. +- **Tool posture (OPEN QUESTION 3).** Without evidence in the repository, the project's tool posture (open-source only, commercial EDA, mixed) is OPEN. The methodology composition must remain valid across posture choices. +- **Design language (OPEN QUESTION 11).** The choice of design language — SystemVerilog, traditional Verilog, Chisel/Spinal HDL, Bluespec — has direct implications for verification methodology (e.g., formal property availability on generated RTL, simulation throughput, the place of unit-level testing at the HDL-source level vs. the generated-RTL level, and the applicability of UVM at the source level). This is OPEN pending XH-1 design-language disclosure. Section A's UVM assumption is conditional on this resolution. +- **Microarchitectural class (OPEN QUESTION 2).** The pipeline class — in-order vs. out-of-order, superscalar width, depth of speculation, branch-prediction aggressiveness — materially changes the cost-benefit of formal vs. dynamic and the complexity of pipeline flush / replay verification. The methodology composition must accommodate both extremes. +- **Power architecture (OPEN QUESTION 7).** Whether the core supports clock gating, power gating, retention, and/or DVFS is OPEN. The verification methodology must accommodate each combination. Assuming any specific power feature is invalid without evidence. Power intent (UPF/CPF) integration is conditional on this. +- **Target implementation class (OPEN QUESTION 12).** Whether XH-1 targets ASIC tape-out or FPGA-only deployment fundamentally changes the verification methodology (DFT/ATPG scope, post-silicon layer, sign-off criteria). Treated here as an assumption (ASIC) pending resolution. + +## 128-Core Scalability + +Replication does not change which verification techniques apply to a single core, but it changes their cost-benefit and their composition. Throughout this section, the two cost quantities from the Background are kept distinct: **defect-escape cost** (multiplied by replication count under the homogeneity assumption) versus **verification effort cost** (paid against one RTL). + +- **Core testbench scalability.** A core testbench written for a single instance must be instantiable inside a multicore wrapper without modification. This implies the testbench interfaces to the core through a standard transaction-level interface (or at least a clearly defined boundary), not through global signals. +- **Cost-benefit of formal at replication.** Per-block formal verification has high fixed engineering cost (constructing the proof), but under bit-identical replication, the *re-proof* cost against another elaborated instance of the same RTL is near zero: once a control property is proven, re-running the proof against a re-elaborated instance of the same source is mechanical. This is a property of the proof, not of the underlying effort. If the source is parameterized and elaborated with per-instance parameters that differ, the proof must be re-established for any parameter set that differs. The defect-escape cost argument is the dominant benefit: a proven property that prevents a class of bug multiplies that benefit by the replication count. +- **Cost-benefit of dynamic at replication.** Dynamic simulation at full coverage also has its verification effort paid against one RTL, but full per-instance coverage at 128 instances requires running 128 instances through the same stimulus, which is a real additional compute cost (not strictly linear in wall-clock because the simulator elaborates them in the same run, but real in host compute and memory). The trade-off is a primary argument for the hybrid approach at 128-core scale, and the practical mitigation is sampling, not full per-instance coverage. +- **Workload scaling.** Running the full core testbench in a 128-core environment is generally unnecessary; sampling (verify a subset of cores under full stimulus, sweep register state across all 128) is a more efficient pattern. The sampling strategy is a multicore concern but must be supported by the core testbench's reset/configuration interface. The statistical argument for the chosen sampling fraction (defect-escape probability vs. simulation cost) is a multicore-layer deliverable; the core testbench must support the interface. +- **Coverage aggregation.** Per-core coverage data must be aggregated across the 128 instances. The core testbench should emit coverage in a form that is identifiable by instance index. +- **Debug bandwidth.** 128 cores each generating waveform/log data is impractical. The core testbench must distinguish between detailed-trace mode (one core, full visibility) and lightweight-trace mode (128 cores, summary only). +- **Compile and elaboration cost.** Elaboration of a 128-core testbench can dominate runtime. The core testbench should be elaborable standalone and in a multicore wrapper without recompiling the core RTL. Elaboration-time behavior (reset, clock tree, memory subsystem) can differ between 1-core and 128-core contexts; the core testbench must be valid in both. +- **Testbench memory footprint.** 128 instances of a complex scoreboard and 128 reference models can dominate host memory. Memory-aware testbench design (shared reference models, deferred-check queues, summary-only trace mode) is required. This is acknowledged here and detailed in the multicore layer. +- **Reference-model parallelism.** Spike and Sail are single-threaded by default. Running 128 reference models in lockstep with 128 RTL instances is a significant engineering problem. The recommended pattern is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling. This is a multicore-layer design constraint that the core testbench interface must support. +- **Symmetry assumption revisit.** The replication-homogeneity assumption is load-bearing throughout this section. Heterogeneous cores (e.g., big/little, mixed application/accelerator, per-core ISA-subset configuration) invalidate the per-die marginal-cost argument and require per-variant verification, including per-variant testbenches, per-variant coverage, and per-variant formal. The methodology should be designed so that the additional cost of a variant is linear in the number of variants, not linear in 128 × variants. + +## Performance Considerations + +The following are framework-level considerations. INSUFFICIENT EVIDENCE exists to provide numerical throughput estimates without XH-1-specific RTL data, simulator choice, and toolchain selection; any specific number is unsourced and is therefore not asserted here. + +- **Testbench throughput.** Dynamic simulation throughput determines how many random seeds can be run per unit wall-clock. Constrained-random at the core level is sensitive to simulator choice, RTL abstraction level, and coverage instrumentation. ASSUMPTION: representative numbers are available in vendor documentation (commercial simulators) and in the open-source Verilator and cocotb benchmark literature, but are workload-dependent and cannot be quoted as typical without measurement on the XH-1 DUT. No specific throughput range is asserted. +- **Compiler/optimizer.** Coverage instrumentation slows simulation. Production coverage runs and fast-turnaround debug runs should be separable. +- **Co-simulation overhead.** PROPOSAL: Spike co-simulation overhead per committed instruction is non-trivial in co-simulation in the strict sense (two-engine lockstep); under the reference-model-checking pattern adopted in Section D, the overhead is a foreign-language call per synchronization point. Synchronization strategy (per-instruction vs. batched) is a tradeoff between debug granularity and throughput. At 128 instances, the cost of a per-instruction synchronization across all instances is prohibitive; batched or sampled synchronization is the practical pattern. +- **FPGA prototype throughput.** FPGA prototype throughput is set by Fmax, which is determined by LUT count, routing, and the memory subsystem. Quoted MHz figures for an XH-1 core on a specific FPGA platform are not asserted here; INSUFFICIENT EVIDENCE without platform and synthesis data. +- **Regression wall-clock budget.** A nightly regression that runs N seeds on a single core must fit within the available compute budget. Cloud burst capacity is typical. + +INSUFFICIENT EVIDENCE to provide numerical throughput estimates without XH-1-specific RTL data and toolchain selection. + +## Area Considerations + +- **On-die verification infrastructure.** XH-1 is described as a 128-core design. The verification *methodology* does not directly consume die area, but decisions made during verification (e.g., embedding assertions in RTL, adding observability muxes for trace, adding per-core DFT structures) do. +- **Trace and debug infrastructure.** Core-level verification informs the design of trace, performance counters, and debug-transport blocks. Larger trace infrastructure costs area; the verification program must justify the cost in terms of debug productivity and post-silicon coverage. +- **DFT (Design-for-Test) compatibility.** Scan chains, BIST, and MBIST are typically owned by a separate DFT program. Core verification should ensure that the core's DFT structures do not alter functional behavior in mission mode. If XH-1 is FPGA-only (OPEN QUESTION 12), DFT is out of scope. +- **Replication-driven area optimization.** If the core's area is a major constraint (which is likely at 128 cores on a single die), the verification plan must protect against defects that would force a re-spin with re-spinned core area. + +INSUFFICIENT EVIDENCE to quantify area without access to XH-1's core area budget or process node. + +## Power and Energy Considerations + +The following considerations apply **if and only if** XH-1 implements the corresponding power features. Each is OPEN QUESTION 7-conditional. + +- **Power-aware verification (conditional).** If the core supports power gating or retention, the core testbench must be able to drive representative switching activity to enable power estimation, and the testbench must verify that power-gated domains behave correctly (controlled wake, retention restore, ordering with respect to coherence traffic). UPF/CPF power intent integration with the simulator is the standard mechanism. +- **Energy proportionality (conditional).** If the core supports clock gating at the instruction or block level, the testbench should verify that idle and active transitions behave correctly at the architectural interface (no observable state loss in idle, correct wake on interrupt). +- **Thermal and DVFS interactions (conditional).** If XH-1 supports DVFS at the core level, voltage/frequency transitions and their effects on memory coherence, pipeline flush behavior, and CSR access must be verified. This is a cross-cutting concern between core and system verification. +- **Power estimation accuracy.** Pre-silicon power estimation depends on workload realism. A core testbench that only runs synthetic random code will produce misleading power numbers. + +No power feature is assumed. Verification provisions for any power feature must be activated only when OPEN QUESTION 7 is resolved. + +## Implementation Considerations + +- **Methodology composition.** The project must select one of the alternative strategies (or a hybrid) and codify it in the verification plan. +- **Toolchain selection.** Open-source (Verilator, cocotb, SymbiYosys, `riscv-formal`, `riscv-dv`, Spike, RISCV-ISAC, GTKWave) vs. commercial (Synopsys VCS, Cadence Xcelium, Siemens Questa, JasperGold, VC Formal, FormalPro, Formality, Conformal, Palladium, Veloce, ZeBu) is a strategic decision. The choice cascades through licensing, hiring, training, and CI. These categories are not equivalent in role: Verilator and cocotb are simulation/cosim frameworks; SymbiYosys is a formal-verification flow; `riscv-formal` is an ISA-level formal property set; `riscv-dv` is a stimulus generator; RISCV-ISAC is an instruction-stream coverage analyzer for the architectural test framework; commercial emulators are hardware-assisted verification platforms distinct from FPGA prototyping. The choice within each role is independent. +- **Reference model choice.** Spike (C++, instruction-accurate) vs. Sail (first-order imperative ISA-specification language with a RISC-V model) vs. a custom golden model. Spike is the most common reference; the project must document any corner-case behaviors of the chosen reference against the RISC-V specification. +- **Coverage database.** Coverage data are stored in vendor-specific formats (e.g., Synopsys UCDB / DDC, Cadence IMC, Siemens UCDB) and may be aggregated into a custom backend (e.g., SQLite) for CI. The format is a tool-vendor concern; the aggregation backend is a CI concern. These are distinct choices. +- **Test categorization.** Tests must be tagged by domain (base integer, M-ext, A-ext, F/D-ext, privileged, vector, …), by intent (sanity, corner, regression, performance), and by status (passing, known-fail, expected-fail). +- **Seed management.** Random seeds should be reproducible. Seed pools should be partitioned into "core regressions", "block regressions", and "fuzzing". +- **Bring-up sequencing.** Block-level → core-level → multicore → system. The core testbench is the transition point between block-level and system-level work. + +## Sign-Off Criteria + +A core verification sign-off is the point at which the methodology exits the coverage-closure funnel. The following are candidate criteria; **every numeric threshold below is an illustrative starting point and is not a validated target**. The relationship to DRAFT status is restated at each threshold so that the list is internally consistent with the document status header. + +- **Functional coverage.** All defined cover points hit, with cross-coverage exercised. **Illustrative starting threshold (not a validated target):** ≥ 95% cover-point hit, ≥ 90% cross-coverage hit; remaining items must be justified as unreachable or formally proven. Final threshold to be set in the verification plan. +- **Code coverage.** Statement, branch, expression, toggle, and FSM coverage reported; excluded regions justified. **Illustrative starting threshold (not a validated target):** ≥ 95% statement, ≥ 90% branch, with manual exclusion of reset/DFT/clock-gating logic. Final threshold to be set in the verification plan. +- **Assertion coverage.** All SVA / immediate assertions vacuously satisfied or actively proven; assertion cover points hit. Open assertions must be reviewed and dispositioned. +- **Formal closure.** Per-block formal sign-off, with proof closure recorded for each property. **Operational requirement:** every inconclusive formal result must have a documented risk-acceptance disposition (strengthen via abstraction/decomposition, or accept with justification) recorded in the sign-off package. "Inconclusive" is not equivalent to "proven". +- **ISA compliance evidence.** All applicable tests in `riscv-arch-test` pass and produce the architectural test framework's signature. `riscv-tests` (broader collection including `riscv-arch-test`) is run as a regression input. Signed certificate or equivalent record retained. **Prerequisite:** the specific RISC-V ISA, Privileged ISA, Debug, and Profile versions adopted by XH-1 are pinned in the verification plan. +- **Mutation testing.** Periodic injection of known-defect mutations into a frozen RTL revision. **Illustrative starting threshold (not a validated target):** ≥ 80% of injected mutations detected by the testbench + assertions + formal. This number is a starting point and requires empirical validation on the XH-1 DUT. Equivalent-mutant handling (mutations that produce behaviorally identical RTL) must be defined; kill-rate targets are noisy and must be interpreted with caution. Final threshold and operator set to be set in the verification plan. +- **Reference model validation.** Spike or Sail validated against the chosen RISC-V compliance tests before being used as a checker; validation evidence archived. +- **Testbench self-check verification.** Every scoreboard check has at least one negative test that is expected to fail. +- **Reproducibility.** Every sign-off run is reproducible from committed source, seed, and tool version. +- **CDC/RDC closure.** CDC and RDC analysis clean for all core-level crossing structures. +- **Lint clean.** Lint clean at the policy-defined severity level. +- **Reset verification.** All reset domains and reset sequences verified; reset recovery verified. + +## Verification Considerations + +Core verification is itself subject to verification — the verification of the verification environment: + +- **Testbench self-checking.** All scoreboard checks must be end-of-test verifiable; every check should be paired with at least one negative test that is expected to fail. +- **Reference model validation.** The reference model (Spike, Sail) must be validated against the chosen RISC-V compliance tests before being used as a checker. +- **Coverage sanity.** Cover points must be reviewed to ensure they are not trivially self-firing. +- **Mutation testing.** A subset of RTL defects should be intentionally injected to confirm the testbench catches them. See `research/10-verification/fault-injection.md`. Equivalent-mutant handling must be defined. +- **Coverage baseline stability.** Coverage baselines drift as RTL changes; a baseline-management policy (when a baseline is re-established, by whom, with what approval) is required. +- **Tool qualification.** If the verification tools are used in any safety-relevant context, tool qualification (per ISO 26262, IEC 61508, DO-254, or similar) is required. The applicability of these standards to XH-1 is OPEN QUESTION 8. Tool qualification is a verification-of-verification concern, not only a sign-off checkbox. +- **Version control and baselines.** Every test run must be reproducible from committed source. Coverage baselines must be versioned. +- **Testbench verification at 128-instance scale.** The testbench itself has bugs. Verification of the testbench (negative tests, mutation of the testbench, equivalence of testbench revisions) is part of the methodology. At 128 instances, the testbench's own correctness becomes a non-trivial verification problem and is acknowledged here as a multicore-layer concern that the core testbench interface must support. + +## Software Considerations + +- **Toolchain integration.** The core verification environment must integrate with the software toolchain (compiler, assembler, linker, Spike, debug server) to enable reference-model checking / co-simulation and to run real programs. +- **Firmware.** Firmware that runs on the core (boot code, exception handlers, interrupt dispatchers) is itself verification target; the core testbench should support loading and running firmware images. +- **Operating system implications.** A core that supports an OS (Linux, RTOS) must be verified with syscall-heavy workloads. The core testbench should provide OS-aware run modes. +- **Debug and trace protocols.** The core's debug-transport module (per the RISC-V Debug specification) must be exercised. JTAG or cJTAG interactions with the core's debug module are typically a core-level concern. +- **Performance analysis tools.** Tools that analyze trace data, performance counters, and profile information should be able to consume the artifacts the core testbench produces. +- **Software developer ergonomics.** The verification team's interfaces (test writers, coverage reviewers) should be ergonomic. A "developer can run a single test and see results in under a minute" goal is a useful operational target. + +## Security Considerations (Out of Scope, but Flagged) + +**ASSUMPTION (out-of-scope for this revision):** Transient-execution / side-channel verification (e.g., Spectre-class, Meltdown-class, and constant-time verification) is treated as out of scope for the XH-1 core verification layer for this revision. For a 128-core design with privilege boundaries, this is a real and high-risk concern. This is flagged as a high-risk gap and as a candidate for a follow-up document. The current methodology is sufficient for functional and microarchitectural correctness but does not, by itself, provide security assurance. + +## Proposed Direction + +PROPOSAL — A **hybrid methodology** for XH-1 core verification, with the following structure: + +1. **Dynamic simulation** as the primary verification engine, with the framework (UVM vs. lightweight) gated on OPEN QUESTION 3 (tool posture), OPEN QUESTION 4 (team size and timeline), and the design-language choice (OPEN QUESTION 11). + - If UVM: pair with `riscv-dv` as the instruction-stream generator and Spike or Sail as the reference model. + - If lightweight: pair `riscv-dv` (or a custom generator) with cocotb / Verilator and Spike, with explicit acknowledgment of Verilator's reduced debug, SVA, and coverage support relative to commercial simulators. +2. **Formal verification** applied to control-dominated blocks: fetch, decode, CSR file, MMU/TLB, exception/interrupt delivery, and the debug module. Coordinate with `research/10-verification/formal-verification.md` for tool selection. Apply `riscv-formal` as an ISA-level formal layer subject to the applicability evaluation in Section I. Sequential equivalence checking between major RTL revisions is a complementary activity. +3. **Reference-model checking against a RISC-V reference model** (Spike or Sail) as the default, with **co-simulation in the strict sense** retained as a sampled cross-check (per Section D). Coordinate with `research/10-verification/isa-compliance.md`. Pin the RISC-V ISA, Privileged ISA, Debug, and Profile versions. At 128 instances, use a shared reference model with a deferred-check queue or sampled synchronization. +4. **Directed tests** for XH-1 microarchitectural corner cases that random cannot reach economically. The corner cases to target are conditional on the in-order / out-of-order answer in OPEN QUESTION 2. +5. **FPGA prototyping** for throughput-driven verification and as a substrate for fault injection and multicore bring-up. Commercial emulation is a conditional alternative based on licensing posture (OPEN QUESTION 3) and workload. +6. **Regression infrastructure** with version-controlled seeds, baselines, and coverage data. Coordinate with `research/10-verification/regression.md`. +7. **Coverage closure** as an explicit, tracked funnel with the sign-off criteria in the Sign-Off Criteria section, including baseline-management policy. Numeric thresholds are illustrative starting points and are not validated targets; final thresholds are deferred to the verification plan. +8. **Reusable testbench architecture** so the core testbench composes inside the multicore testbench without modification, supporting both the per-die cost argument and the sampling strategy. The interface must support shared reference models, summary-only trace mode, and per-instance coverage identification. +9. **Mutation testing** as a periodic check that the testbench is still doing useful work, with explicit equivalent-mutant handling. Coordinate with `research/10-verification/fault-injection.md`. +10. **Standard core-verification adjuncts:** lint, CDC/RDC analysis, X-propagation handling, reset verification, and (conditional on OPEN QUESTION 7) UPF/CPF power-intent integration. + +The hybrid is proposed; the specific framework within the hybrid (UVM vs. lightweight) is gated on OPEN QUESTION 3, OPEN QUESTION 4, and OPEN QUESTION 11, and the power-feature sub-portion is gated on OPEN QUESTION 7. The implementation-class sub-portion (DFT/ATPG scope, post-silicon layer) is gated on OPEN QUESTION 12. The document does not pick between alternatives that the OPEN QUESTIONS do not yet support. + +## Confidence + +- **High confidence:** The general methodology composition (dynamic + formal + reference-model checking / co-simulation + directed + prototype + regression + coverage closure, with the standard adjuncts: lint, CDC/RDC, X-propagation, reset) corresponds to the IEEE 1800.2-2017 UVM standard, the IEEE 1800-2017 SystemVerilog assertion subset, and the RISC-V International published ISA / Privileged / Debug / Profile specifications, and is broadly accepted in industrial practice. The references supporting each component are cited at the point of use. +- **High confidence (conditional):** The 128-core scaling argument on defect-escape cost (an undetected defect is replicated 128-fold in silicon) is a direct consequence of the replication-homogeneity assumption and does not require empirical data. The argument about formal re-proof cost under bit-identical replication (near-zero marginal cost to re-run a proof against another elaborated instance of the same source; engineering cost of constructing the proof is paid once) is also a direct consequence of the assumption, with the parameter-set caveat acknowledged. The argument about dynamic simulation at full per-instance coverage requiring real additional compute is similarly structural. +- **Medium confidence:** Specific tool selections, sign-off numeric thresholds, and area/power tradeoffs, given the absence of repository evidence on the OPEN QUESTIONS. +- **Low confidence (OPEN):** XH-1's tool licensing posture, core microarchitecture details, ISA profile selection, team size, design language, power architecture, and target implementation class. These drive the specifics within the proposed hybrid. + +## Open Questions + +1. What RISC-V profile does the XH-1 core implement, and what is the specific version pin for the Unprivileged ISA, Privileged ISA, and Debug specification? The reference model (Spike variant, Sail configuration) and the compliance test scope follow from this choice. The RISC-V International profile document should be consulted at toolchain selection to determine the exact profile designation adopted (e.g., a 64-bit XLEN core targeting an RVA application/supervisor profile, or a 32-bit XLEN core targeting a different profile track); this document does not assert a specific profile name as established fact. +2. Is the XH-1 core in-order or out-of-order? The verification plan complexity differs substantially. The instructions explicitly disallow assuming either without repository evidence. +3. What is XH-1's toolchain posture (open-source only, commercial EDA, mixed)? +4. What is the team size and timeline? Methodology cost scales with team capacity. +5. What is the core's microarchitectural contract — what blocks are in scope for unit verification, and what is the boundary between core and multicore? See `research/10-verification/unit-verification.md` and `research/10-verification/multicore-verification.md`. +6. Are the 128 cores architecturally identical, or are there variants (big/little, mixed application/accelerator, per-core ISA-subset configuration)? Heterogeneity changes the testbench architecture and invalidates the per-die cost-leverage argument. +7. What is the core's power architecture — is clock gating, power gating, retention, or DVFS at the core level, and what are the state-retention semantics? +8. What is XH-1's relationship to functional safety standards (ISO 26262, IEC 61508, DO-254, or none)? +9. How will the XH-1 core's trace, debug, and performance-counter interfaces be designed, and how will core verification consume them? +10. What is the expected defect rate target at tape-out, and how does the verification program demonstrate closure against that target? +11. What is the XH-1 design language (SystemVerilog, Verilog-2001, Chisel/Spinal HDL, Bluespec), and does verification target the HDL source level, the generated-RTL level, or both? +12. What is XH-1's target implementation class — ASIC tape-out, FPGA-only deployment, or both? This determines DFT/ATPG scope, post-silicon validation depth, and sign-off criteria. +13. What physical implementation tool flow (synthesis, PnR, timing) is envisioned, and what are the timing-closure constraints (target clock frequency, slack budget)? + +## Sources + +- IEEE 1800.2-2017, *Standard for Universal Verification Methodology*. Defines UVM. The Accellera UVM reference implementation is distributed under the Apache License 2.0; the IEEE 1800.2-2017 standard document is an IEEE publication. +- IEEE 1800-2017, *Standard for SystemVerilog — Unified Hardware Design, Specification, and Verification Language*. Defines SVA and immediate assertions. +- RISC-V International, *The RISC-V Instruction Set Manual, Volume I: Unprivileged Architecture*. **Version pin: OPEN** — to be selected at toolchain selection and recorded in the verification plan. +- RISC-V International, *The RISC-V Instruction Set Manual, Volume II: Privileged Architecture*. **Version pin: OPEN** — to be selected at toolchain selection and recorded in the verification plan. +- RISC-V International, *RISC-V External Debug Support*. **Version pin: OPEN** — to be selected at toolchain selection and recorded in the verification plan. +- RISC-V International, *RISC-V Profiles*. **Version pin: OPEN** — the specific profile designation adopted by XH-1 is OPEN until OPEN QUESTION 1 is resolved. The published RISC-V profile document should be consulted directly at toolchain selection; this document does not assert a specific profile name. +- Symbiotic EDA / YosysHQ, *SymbiYosys* documentation (open-source formal verification flow). +- `riscv-formal` framework (Verilog, BSD-licensed), maintained at github.com/YosysHQ/riscv-formal. +- `riscv-dv` instruction-stream generator, maintained at github.com/google/riscv-dv. +- `riscv-tests` and `riscv-arch-test`, maintained by RISC-V International. `riscv-arch-test` is the architectural conformance test framework contained within the broader `riscv-tests` repository. +- Spike (the official RISC-V ISA simulator), maintained at github.com/riscv-software-src/riscv-isa-sim. +- Sail RISC-V formal specification, maintained at github.com/rems-project/sail. Sail is a first-order imperative ISA-specification language; the RISC-V model is generated from the RISC-V ISA formal definition. +- RISCV-ISAC, instruction-stream coverage analyzer for the RISC-V architectural test framework, maintained by RISC-V International. + +Quantitative throughput, area, power, and timing claims are not asserted in the body. The XH-1 repository documents referenced (under `research/10-verification/`) are placeholders ("SOON") and do not provide factual content to cite. No measurements, benchmarks, processor capabilities, URLs outside the repositories named above, or experimental results have been invented. diff --git a/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round4.md.json b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round4.md.json new file mode 100644 index 0000000..c2c8f4c --- /dev/null +++ b/research/.xh1/runs/20260826T101823Z/research_10-verification_core-verification.md.round4.md.json @@ -0,0 +1,353 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787739795-j5q2quc3b7bB2h3T64og","object":"chat.completion","created":1787739795,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"# Core Verification\n\n## Status\n\nDRAFT — Research framework established; all claims are explicitly classified by epistemic status (FACT with citation, ASSUMPTION, PROPOSAL, RECOMMENDATION, OPEN QUESTION, or INSUFFICIENT EVIDENCE). Concrete sign-off criteria, tool selection, and quantitative thresholds are deferred until the OPEN QUESTION inputs are resolved. The numeric thresholds appearing in the Sign-Off Criteria section are illustrative starting points for the verification plan and are not validated targets. Every threshold below is restated with this caveat at the point of use so that the sign-off list is internally consistent with this status.\n\n## Abstract\n\nThis document investigates verification strategies for the XH-1 core, defined here as a single processing core instance within a custom 128-core RISC-V processor. Core verification is the layer of the verification methodology stack concerned with demonstrating that one core conforms to its architectural specification, microarchitectural contract, and integration assumptions prior to and independent of multi-core interactions. For a 128-core design, the core is the unit of replication; defects undetected at this layer replicate 128-fold across the die, amplifying area, power, yield, and debug cost. This document frames the research questions, surveys existing approaches (simulation, formal verification, hardware-assisted verification, FPGA prototyping, emulation), analyzes their applicability to a custom RISC-V core, and identifies the gaps that the XH-1 project must close given that the existing related documents in `research/10-verification/` are placeholders (\"SOON\").\n\n## Research Question\n\nWhat verification methodology, toolchain, environment abstraction, coverage model, and sign-off criteria are appropriate for verifying a single XH-1 core such that the methodology:\n\n1. Provides evidence that the core implements the RISC-V ISA subset selected for XH-1 (see `research/10-verification/isa-compliance.md`).\n2. Provides evidence that the core's microarchitecture meets timing, area, and power targets for a design that will be replicated 128 times.\n3. Scales economically — verification effort, compute cost, and human review time must not grow linearly with the 128-core replication factor.\n4. Produces artifacts (testbenches, coverage models, formal proofs, regression suites) that compose cleanly with the surrounding verification layers (unit, multicore, system, post-silicon).\n5. Is implementable given that the related XH-1 documents in area 10 are not yet populated, indicating a greenfield verification program.\n\n## Background\n\nVerification of a processor core is conventionally decomposed into:\n\n- **Unit verification** — verification of individual blocks (ALU, FPU, register file, branch predictor, fetch/decode structures, load-store unit, replay logic, CSR file, MMU/TLB, caches). See `research/10-verification/unit-verification.md`.\n- **Core verification** — verification of the integrated core as a single design under test (DUT), exercising instruction-stream and exception behavior at the architectural interface, plus internal microarchitectural invariants.\n- **Multicore verification** — coherence, consistency, interconnect, and shared-resource correctness across multiple core instances. See `research/10-verification/multicore-verification.md`.\n\nFor XH-1, the core is the replication unit. The cost model must distinguish two quantities that this document often refers to together as \"cost\":\n\n- **Defect-escape cost (multiplied by replication).** The downstream cost of an undetected defect (silicon debug, re-spin, field failure, yield loss). Under the replication-homogeneity assumption, an undetected core defect is replicated 128-fold in silicon, multiplying its downstream impact.\n- **Verification effort cost (paid once).** The engineering cost of running the testbench, discharging formal properties, maintaining the regression, and writing tests. This is paid against one RTL; the fix, once made, is replicated.\n\nThe phrase \"every bug caught at the core layer is caught 128 times\" is shorthand for the first quantity: every defect that escapes core verification multiplies its downstream cost by the replication count. It is not a claim that the verification effort itself runs 128 times — the testbench runs against the RTL once.\n\n**ASSUMPTION (replication homogeneity):** The 128 XH-1 cores are architecturally identical instances of a single RTL core produced by elaboration of the same source. Heterogeneous cores, per-core configuration, or runtime variants invalidate the replication-leverage argument on defect-escape cost and require per-variant verification. This is treated as an assumption because no repository evidence contradicts it; OPEN QUESTION 6 below revisits it. The \"per-instance marginal cost of zero\" claim for proven formal properties holds for bit-identical replicated RTL and refers to the marginal cost of *re-running* a proof against another elaborated instance of the same source — not to the engineering cost of constructing the proof. If the source is parameterized and elaborated with per-instance parameters that differ, the proof must be re-established for any parameter set that differs, and that re-establishment cost is not zero.\n\n**ASSUMPTION (target implementation class):** XH-1 targets ASIC tape-out. FPGA-only deployment would change the methodology (no DFT/ATPG, different sign-off criteria, no post-silicon bring-up in the same sense). This assumption is conditional and is revisited in OPEN QUESTION 12.\n\n**ASSUMPTION (verification language availability):** The verification language of the testbench is the same as or interoperable with the design language. This is revisited in OPEN QUESTION 11.\n\nCore verification draws on several formal and semi-formal techniques:\n\n- **Dynamic simulation** with constrained-random or directed tests, against an ISA reference model, with functional coverage.\n- **Static (formal) verification** — model checking (bounded model checking, k-induction, IC3/PDR), equivalence checking (combinational and sequential), and property proving against the RTL.\n- **Hardware-assisted verification** — FPGA prototyping and emulation (commercial emulators, open-source flows such as FireSim on Amazon EC2 F1), trading compile time for runtime speed.\n- **Co-simulation** — two simulation engines running in lockstep (e.g., an RTL simulator and an architectural reference-model engine such as Spike) synchronized at defined events. This is distinct from **reference-model checking**, in which the reference model is integrated as a transactor within a single simulator; the latter is a common industrial pattern and is more accurately described as reference-model checking, not co-simulation in the strict sense.\n- **Waveform and assertion-based methodologies** — SystemVerilog Assertions (SVA), PSL, custom monitors, end-of-test checkers.\n\nCoverage is a central concern. Functional coverage, code coverage (line, branch, expression, toggle, FSM), and cross-coverage of architectural state are the standard dimensions. For a custom ISA implementation, the relationship between ISA compliance coverage and microarchitectural coverage must be defined explicitly.\n\nStandard core-verification concerns that must be addressed in addition to instruction-stream behavior:\n\n- **Lint** — static analysis of RTL style, common bug patterns, and synthesizability (e.g., Verilator lint, commercial lint tools). A prerequisite to dynamic and formal work.\n- **Clock-domain crossing (CDC) and reset-domain crossing (RDC)** — analysis of signals crossing asynchronous clock or reset domains. Typically owned by a separate methodology but the core must be CDC/RDC-clean for integration.\n- **X-propagation** — controlled handling of unknown ('X) propagation in simulation, using X-pessimism removal or X-init optimization, to avoid false coverage from masked X's.\n- **Reset verification** — power-on reset, software reset, debug reset, partial reset, and reset-domain interaction.\n- **Power intent** — if power features are present (see OPEN QUESTION 7), UPF/CPF power intent formats integrate with simulation to model power-gated, retention, and isolation behavior.\n- **DFT/ATPG interaction** — scan, BIST, and MBIST structures must not alter mission-mode behavior; their interaction with the core testbench is a coordination point.\n\nThese are not optional; their absence in a verification plan is a known gap in pre-silicon methodology.\n\n## Existing Approaches\n\nThe following approaches are established in industry and academic practice. Their applicability to XH-1 is analyzed; none is recommended in isolation.\n\n### A. Constrained-Random Simulation with UVM\n\n**FACT:** Universal Verification Methodology (UVM) is standardized as IEEE 1800.2-2017, *Standard for Universal Verification Methodology* [IEEE 1800.2-2017], and is widely used in industrial block- and core-level processor verification for constrained-random stimulus generation, reference model integration, self-checking scoreboards, functional coverage, and phasing. The Accellera UVM reference implementation, which is the source code that vendors and users actually consume, is distributed under the Apache License 2.0; the IEEE 1800.2-2017 standard document itself is an IEEE publication and is not freely redistributable. The two are distinct artifacts and this distinction is load-bearing when characterizing \"open-source viability\" below.\n\n**ASSUMPTION (conditional on design language):** UVM is applicable only if the XH-1 design language supports SystemVerilog class constructs — either the source is SystemVerilog, or the verification environment targets SystemVerilog regardless of source language. If the design language is Verilog-2001 only, or if the project commits to a Chisel/Spinal HDL or Bluespec-only flow, UVM is not directly applicable at the HDL-source level and an alternative testbench harness is required. This assumption is revisited in OPEN QUESTION 11.\n\n**PROPOSAL (applicability to XH-1):** Apply UVM as the primary dynamic verification environment for XH-1 core-level testing, paired with a chosen ISA reference (Spike or Sail). Sequence generation over the RISC-V ISA state space; reference model in co-simulation; scoreboard for architectural state comparison per committed instruction; explicit cover points for instruction × operand-mode × privilege × exception interactions.\n\n### B. Directed and Hand-Written Tests\n\n**FACT:** Constrained-random alone does not reach corner-case coverage of complex pipelines, corner-case CSR interactions, or rare exception sequences. Industry practice supplements with directed tests written by architects and designers.\n\n**PROPOSAL (applicability to XH-1):** A XH-1-specific directed test suite for pipeline hazard combinations specific to the XH-1 microarchitecture, corner cases in the chosen RVA profile subset, reset, wake, low-power entry/exit, and debug transport (if present at core level).\n\n### C. Formal Property Verification\n\n**FACT:** Bounded model checking, k-induction, and IC3/PDR are applied to control-dominated blocks: fetch, decode, CSR, MMU/TLB, branch predictor, retirement, and exception delivery. Commercial model checkers include Cadence JasperGold, Synopsys VC Formal, and Siemens FormalPro. Open-source flows include SymbiYosys (SBY), based on the Yosys synthesis and formal toolchain. Equivalence checking is a distinct formal category: tools include Synopsys Formality and Cadence Conformal, used to verify that two RTL representations (e.g., pre- and post-synthesis, or pre- and post-retiming) are functionally equivalent. Equivalence checkers are not model checkers and are not used for property proving.\n\n**PROPOSAL (applicability to XH-1):** Per-block formal sign-off for the control-dominated blocks listed above, end-to-end pipeline flushing, exception priority, and CSR read/write semantics. Formal proofs of privileged ISA semantics (e.g., `mstatus`, `sstatus`, `satp`, `mip`/`sip` behavior) provide a high-confidence anchor that random tests cannot match in isolation. Sequential equivalence checking between major RTL revisions is a separate, complementary activity.\n\n**ASSUMPTION (RISC-V privileged-ISA versioning):** The CSR semantics used as formal-proof targets correspond to the specific RISC-V Privileged ISA version adopted by XH-1. The unprivileged ISA, privileged ISA, and Debug specification each have versioned releases tracked by RISC-V International; the verification plan must be pinned to those versions. The specific version is OPEN until resolved (see OPEN QUESTION 1 and the Sources section).\n\n**PROPOSAL (applicability to XH-1):** Coordinate with `research/10-verification/formal-verification.md` for tool selection (commercial model checkers vs. SymbiYosys / SBY) and with `research/10-verification/isa-compliance.md` for the privileged-ISA version.\n\n### D. Reference-Model Checking and Co-Simulation\n\n**FACT:** Comparing DUT architectural state against Spike (the official RISC-V ISA simulator, C++ implementation) or Sail (a first-order imperative language designed for ISA specification, with a RISC-V model generated from the RISC-V ISA formal definition) per instruction is a widely deployed pattern. The terminology matters:\n\n- **Co-simulation (strict sense):** two independent simulation engines (e.g., an RTL simulator and Spike) running in lockstep, synchronized at defined events. The DUT may run ahead of the reference using a deferred-check queue, with results compared at synchronization points.\n- **Reference-model checking:** a single simulator that incorporates the reference model as a transactor or foreign-language call. This is a common industrial pattern but is not co-simulation in the strict sense.\n\n**PROPOSAL (applicability to XH-1):** Required for ISA-compliance evidence. See `research/10-verification/isa-compliance.md`. The specific architecture proposed is **reference-model checking within a single simulator** (Spike invoked as a transactor from the UVM or lightweight harness) as the default for day-to-day runs, with **co-simulation in the strict sense** retained as a cross-check on a sampled subset of tests to detect synchronization artifacts that a single-simulator harness could mask. Reference model parallelism is a 128-core scaling concern: Spike and Sail are single-threaded by default; running 128 reference models in lockstep with 128 RTL instances is itself a significant engineering problem. The pattern of choice is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling (multicore-layer decision) with the core testbench supporting the interface.\n\n### E. FPGA Prototyping and Emulation\n\n**FACT:** FPGA-based prototyping runs the RTL at higher throughput than software simulation, and emulation platforms (commercial: Palladium, Veloce, ZeBu; open-source: FireSim on Amazon EC2 F1) provide a middle ground. The throughput multiplier relative to software simulation is workload- and instrumentation-dependent.\n\n**PROPOSAL (applicability to XH-1):** Use FPGA prototyping to speed up core testbench bring-up by running entire test programs end-to-end, to enable fault-injection campaigns that are infeasible in simulation (see `research/10-verification/fault-injection.md`), and as the natural place to begin validating the multicore layer; core verification is the upstream prerequisite. Commercial emulation is a viable alternative for very large workloads or when observability requirements are relaxed; the decision depends on licensing posture (OPEN QUESTION 3) and workload characteristics. No specific throughput multiplier for the XH-1 DUT is asserted here; INSUFFICIENT EVIDENCE without XH-1-specific RTL and platform data.\n\n### F. Regression Infrastructure\n\n**FACT:** Regression management — version control of testbenches, seeds, baselines, coverage databases — is a load-bearing component of any non-trivial core verification program. See `research/10-verification/regression.md`.\n\n**PROPOSAL (applicability to XH-1):** The XH-1 regression infrastructure must support a single-core nightly run, a multicore run, and a system run with shared seeds where appropriate.\n\n### G. Coverage Closure Methodology\n\n**FACT:** Coverage closure combines functional coverage (defined by the verification plan), code coverage (uncovered RTL), and assertion coverage. Industry practice treats coverage closure as a managed funnel: cover points → cross coverage → corner case identification → directed/random test addition → formal where possible. Coverage baselines drift as RTL changes; a baseline-management policy is required.\n\n**PROPOSAL (applicability to XH-1):** Codify the funnel explicitly in the verification plan with tracked stages, baseline-management policy, and exit criteria per stage.\n\n### H. Assertion-Based Verification\n\n**FACT:** SystemVerilog Assertions (IEEE 1800-2017, *Standard for SystemVerilog*) and immediate assertions encode design intent and serve both as live checkers and as inputs to formal tools.\n\n**PROPOSAL (applicability to XH-1):** Define an SVA library for XH-1 invariants (e.g., \"no retired instruction has a privileged side effect inconsistent with its mode\", \"no speculative architectural state is observable at the architectural interface\").\n\n### I. Open-Source ISA-Level Formal Verification\n\n**FACT:** The `riscv-formal` framework (YosysHQ / Symbiotic EDA, Verilog, BSD-licensed) provides a checkable RISC-V ISA specification encoded as a set of formal properties that are conjoined with a wrapper around the DUT and discharged by an SMT solver. The Kami project (Cambridge) provides a formal RISC-V ISA specification in Coq with a Bluespec front-end. INSUFFICIENT EVIDENCE to assert at the FACT level that `riscv-formal` is \"directly applicable\" to a specific XH-1 microarchitecture and ISA profile without further evaluation: applicability depends on the DUT's memory and CSR interface matching the framework's wrapper assumptions, on the chosen RISC-V profile, and on the unprivileged/privileged ISA versions adopted.\n\n**PROPOSAL (applicability to XH-1):** Evaluate `riscv-formal` as a complement to block-level formal. It targets ISA-compliance evidence with open-source tooling and is particularly relevant given the greenfield status of the XH-1 verification program, subject to the applicability evaluation above.\n\n### J. Open-Source Constrained-Random Generator\n\n**FACT:** Google's `riscv-dv` is an open-source UVM-based constrained-random instruction-stream generator for RISC-V, widely used in industrial and academic flows. The license should be confirmed against the current repository state at the time of toolchain selection; this document does not assert a specific license version.\n\n**PROPOSAL (applicability to XH-1):** Evaluate `riscv-dv` as a generator either within a UVM environment or as a stimulus source for a lightweight harness.\n\n### K. RISC-V Architectural Test Framework\n\n**FACT:** RISC-V International maintains `riscv-arch-test` as the architectural test framework, which is the conformance test suite against which cores are signed off for profile compliance. `riscv-tests` is a related umbrella repository containing `riscv-arch-test` as well as additional tests; the relationship is not \"legacy vs. replacement\" — `riscv-tests` is the broader collection and `riscv-arch-test` is the architectural conformance subset. The two should be treated as distinct but related inputs, both of which produce evidence in the sign-off record. The framework's coverage analyzer, `RISCV-ISAC`, is an instruction-stream coverage analyzer used to confirm that the architectural test signatures cover the required instruction/operand/privilege combinations; it is not a generic functional-coverage database and is not a substitute for the core testbench's own coverage model.\n\n**PROPOSAL (applicability to XH-1):** Use `riscv-arch-test` for the ISA-compliance certificate, with `RISCV-ISAC` analysis of the architectural test signatures as part of the coverage evidence, and `riscv-tests` more broadly as a regression input. The license and current revision of each must be confirmed at toolchain selection.\n\n## Alternative Designs\n\nThe following alternative *strategies* for the core verification methodology itself are candidates for XH-1. They are not mutually exclusive; the project must select a composition.\n\n### Alternative 1: UVM-Centric with Formal Supplements\n\n- **Approach:** UVM as the workhorse for dynamic verification, with formal applied to control-dominated blocks.\n- **Strengths:** Industry-standard methodology, well-understood coverage and sign-off conventions, mature ecosystem; integrates with `riscv-dv` as a generator and `riscv-formal` as an ISA-level formal layer.\n- **Weaknesses:** UVM has a steep learning curve and high setup cost. For a 128-core design, the UVM environment must be designed for reuse to avoid 128× rewrite cost. Requires SystemVerilog (OPEN QUESTION 11). Implies commercial simulator licensing in most deployments, because open-source SystemVerilog simulation support (Verilator) has limited UVM-class support relative to commercial offerings; the open-source viability of this alternative is therefore medium, not high.\n\n### Alternative 2: Lightweight Python/C++ Test Harness with Reference-Model Checking\n\n- **Approach:** A generator produces instruction streams and a C++ testbench (e.g., based on cocotb, Verilator, or a custom framework) compares architectural state against Spike.\n- **Strengths:** Lower setup cost than UVM. Excellent generator productivity. Easy integration with regression infrastructure. Open-source-friendly. `riscv-dv` can serve as the generator. Verilator-based flows are not bound to SystemVerilog UVM at the source level.\n- **Weaknesses:** Verilator's debug, SVA, and functional-coverage support are not equivalent to commercial simulators; cocotb+Verilator is therefore not a drop-in replacement for UVM, and the trade-off should be evaluated against the XH-1 debug and coverage requirements. Coverage closure tooling is less standardized than the UVM vendor ecosystem. The reference-model-checking pattern (Section D) is the natural fit for this harness.\n- **Application to XH-1:** Reasonable choice if XH-1 has a small team, an open-source toolchain preference, or a constraint on commercial EDA licensing.\n\n### Alternative 3: Formal-First\n\n- **Approach:** Formal proofs form the primary verification evidence; dynamic simulation is used for reach and to validate properties that formal cannot discharge.\n- **Strengths:** Highest assurance per proven property. Strong fit for control logic and CSR behavior. `riscv-formal` provides an open-source ISA-level formal layer subject to the applicability evaluation in Section I. Under the replication-homogeneity assumption, proven properties transfer to all 128 instances at near-zero marginal *re-proof* cost (the proof is re-run, not re-constructed, against another elaborated instance of the same RTL); the engineering cost of constructing the proof is paid once.\n- **Weaknesses:** Pipeline formalization is hard. State-space complexity grows combinatorially for large cores. May not reach proof closure on timing-agnostic models without abstraction. Inconclusive results are not equivalent to failure; risk-acceptance must be tracked.\n\n### Alternative 4: Emulation-Centric\n\n- **Approach:** Skip large-scale software simulation; use FPGA/emulation throughput as the primary engine.\n- **Strengths:** Excellent long-workload capability; native fault-injection support.\n- **Weaknesses:** Poor waveform/observability; slow compile/turnaround; harder to drive corner-case stimulus; debug cycle is slow.\n\n### Alternative 5: Hybrid (PROPOSAL — See Proposed Direction)\n\nA combination of a dynamic framework (Alternative 1 or Alternative 2) for stimulus and reach, formal supplements (Alternative 3, including `riscv-formal` where applicable) for control logic, and FPGA prototyping/emulation for throughput and multicore bring-up.\n\n## Comparison\n\nThe comparison is qualitative and is intended to make the trade-offs explicit, not to produce a numeric ranking. INSUFFICIENT EVIDENCE exists to rank absolute quality without XH-1-specific RTL data, tool licensing posture, and team size. The Hybrid column is a composition of the others and is not an independent point of comparison; the table is structured to show how each pure strategy performs on the dimensions, with the Hybrid column indicating how the strengths compose.\n\n| Dimension | UVM-Centric | Lightweight Harness | Formal-First | Emulation-Centric | Hybrid |\n|---|---|---|---|---|---|\n| Setup cost | High | Low | High | Medium | High |\n| Per-test runtime | Slow | Slow–Medium | N/A (proof time) | Fast | Mixed |\n| Corner-case reach | Strong (with seeds) | Strong (with fuzzing) | Strong (when discharged) | Weak–Medium | Strong |\n| Assurance on control logic | Medium | Medium | High | Low–Medium | High |\n| Debug ergonomics | Good (waveform, logs) | Good (caveat: Verilator/cocotb debug support not equivalent to commercial) | Strong counterexamples | Weak | Mixed |\n| Open-source viability | Medium (UVM standard is IEEE-published; UVM reference implementation is Apache 2.0; simulators are usually commercial) | Strong | Medium | Medium | Medium |\n| Fit for 128-core reuse | High (if designed so) | High (if designed so) | High (under replication homogeneity) | High | High |\n| Pre-tape-out risk reduction | Strong | Medium–Strong | Strong (partial; inconclusive ≠ proven) | Medium | Strong |\n\n## Advantages\n\nThe advantages of investing in a robust core verification methodology for XH-1:\n\n- **Replication leverage on defect-escape cost.** Under the replication-homogeneity assumption, an unfixed defect in the core RTL is replicated 128-fold in silicon, multiplying its downstream (silicon-debug, re-spin, field-failure, yield-loss) cost. Verification effort itself is paid once against one RTL; the defect, if found, is fixed once; the fixed RTL is then replicated. The benefit of core-level detection is the avoidance of multiplied post-silicon cost.\n- **Cleaner multicore verification.** A well-verified core reduces the multicore verification problem to coherence, consistency, interconnect, and shared resources. See `research/10-verification/multicore-verification.md`.\n- **Foundation for post-silicon.** First-stage silicon bring-up is dramatically simpler when the core is well-understood and the debug interfaces (assertions, traces, scan) are well-designed. Post-silicon validation feeds back into pre-silicon coverage targets; this feedback loop is not detailed in this document and is left to the post-silicon layer.\n- **Reusable testbench assets.** Generator, scoreboard, coverage model, and SVA library can be reused across cores if XH-1 has core variants, or across revisions of the same core.\n- **Formal evidence on critical control.** Privileged ISA semantics are hard to verify by random simulation; formal proofs on these blocks provide a high-confidence floor.\n\n## Disadvantages\n\nCosts and risks of a robust core verification program:\n\n- **Engineering cost.** A UVM-class environment plus formal plus emulation is a substantial program. The claim that it \"can easily exceed the RTL design cost for a single core\" is folkloric and is presented here as a planning consideration, not as a measured fact.\n- **Tooling lock-in.** Heavy UVM usage often implies commercial simulator licensing; this may not match XH-1's tool posture. Formal tool lock-in is similar: commercial model checkers differ in property language and run scripts.\n- **Slow turnaround.** Constrained-random simulation with full coverage at the core level is slow; coverage closure can take many cycles of \"add test, rerun, close hole\". Coverage baselines drift with RTL changes, requiring re-establishment.\n- **Reference model dependence.** Spike, Sail, or any reference may diverge from the RISC-V specification in interpretation of ambiguous or profile-specific corners; the verification program must track the chosen RVA profile carefully and pin the spec version of the reference.\n- **Coverage false confidence.** Hitting a cover point is not the same as proving correctness; the verification plan must distinguish cover and check.\n- **128-core replication pressure.** If the core testbench is not designed to compose with the multicore testbench, redundant effort emerges.\n- **Inconclusive formal ≠ proven.** Formal results of \"inconclusive\" must be risk-accepted, not treated as success.\n- **Equivalent mutants.** Mutation-kill rates are affected by equivalent mutants (mutations that produce behaviorally identical RTL); kill-rate targets are therefore noisy and must be interpreted with caution.\n\n## XH-1 Considerations\n\nXH-1-specific factors that constrain the methodology choice:\n\n- **128-core replication.** The verification plan must be replication-aware: the *testbench* should be agnostic to whether one or 128 cores are instantiated. This implies strong separation of (a) the core DUT, (b) the core testbench, and (c) the multicore wrapper.\n- **Custom design.** No third-party core is being verified; there is no vendor's pre-built verification IP. All environments must be built or assembled.\n- **Repository context.** The related verification documents are all placeholders (\"SOON\"). This is a greenfield verification program; the XH-1 project has the opportunity — and the burden — to define the methodology from first principles.\n- **RISC-V ISA.** The RISC-V ecosystem provides mature reference models (Spike, Sail), compliance / architectural test frameworks (`riscv-arch-test`, `riscv-tests`), and open-source verification tooling (`riscv-formal`, `riscv-dv`, `RISCV-ISAC`). These are the natural anchors.\n- **Tool posture (OPEN QUESTION 3).** Without evidence in the repository, the project's tool posture (open-source only, commercial EDA, mixed) is OPEN. The methodology composition must remain valid across posture choices.\n- **Design language (OPEN QUESTION 11).** The choice of design language — SystemVerilog, traditional Verilog, Chisel/Spinal HDL, Bluespec — has direct implications for verification methodology (e.g., formal property availability on generated RTL, simulation throughput, the place of unit-level testing at the HDL-source level vs. the generated-RTL level, and the applicability of UVM at the source level). This is OPEN pending XH-1 design-language disclosure. Section A's UVM assumption is conditional on this resolution.\n- **Microarchitectural class (OPEN QUESTION 2).** The pipeline class — in-order vs. out-of-order, superscalar width, depth of speculation, branch-prediction aggressiveness — materially changes the cost-benefit of formal vs. dynamic and the complexity of pipeline flush / replay verification. The methodology composition must accommodate both extremes.\n- **Power architecture (OPEN QUESTION 7).** Whether the core supports clock gating, power gating, retention, and/or DVFS is OPEN. The verification methodology must accommodate each combination. Assuming any specific power feature is invalid without evidence. Power intent (UPF/CPF) integration is conditional on this.\n- **Target implementation class (OPEN QUESTION 12).** Whether XH-1 targets ASIC tape-out or FPGA-only deployment fundamentally changes the verification methodology (DFT/ATPG scope, post-silicon layer, sign-off criteria). Treated here as an assumption (ASIC) pending resolution.\n\n## 128-Core Scalability\n\nReplication does not change which verification techniques apply to a single core, but it changes their cost-benefit and their composition. Throughout this section, the two cost quantities from the Background are kept distinct: **defect-escape cost** (multiplied by replication count under the homogeneity assumption) versus **verification effort cost** (paid against one RTL).\n\n- **Core testbench scalability.** A core testbench written for a single instance must be instantiable inside a multicore wrapper without modification. This implies the testbench interfaces to the core through a standard transaction-level interface (or at least a clearly defined boundary), not through global signals.\n- **Cost-benefit of formal at replication.** Per-block formal verification has high fixed engineering cost (constructing the proof), but under bit-identical replication, the *re-proof* cost against another elaborated instance of the same RTL is near zero: once a control property is proven, re-running the proof against a re-elaborated instance of the same source is mechanical. This is a property of the proof, not of the underlying effort. If the source is parameterized and elaborated with per-instance parameters that differ, the proof must be re-established for any parameter set that differs. The defect-escape cost argument is the dominant benefit: a proven property that prevents a class of bug multiplies that benefit by the replication count.\n- **Cost-benefit of dynamic at replication.** Dynamic simulation at full coverage also has its verification effort paid against one RTL, but full per-instance coverage at 128 instances requires running 128 instances through the same stimulus, which is a real additional compute cost (not strictly linear in wall-clock because the simulator elaborates them in the same run, but real in host compute and memory). The trade-off is a primary argument for the hybrid approach at 128-core scale, and the practical mitigation is sampling, not full per-instance coverage.\n- **Workload scaling.** Running the full core testbench in a 128-core environment is generally unnecessary; sampling (verify a subset of cores under full stimulus, sweep register state across all 128) is a more efficient pattern. The sampling strategy is a multicore concern but must be supported by the core testbench's reset/configuration interface. The statistical argument for the chosen sampling fraction (defect-escape probability vs. simulation cost) is a multicore-layer deliverable; the core testbench must support the interface.\n- **Coverage aggregation.** Per-core coverage data must be aggregated across the 128 instances. The core testbench should emit coverage in a form that is identifiable by instance index.\n- **Debug bandwidth.** 128 cores each generating waveform/log data is impractical. The core testbench must distinguish between detailed-trace mode (one core, full visibility) and lightweight-trace mode (128 cores, summary only).\n- **Compile and elaboration cost.** Elaboration of a 128-core testbench can dominate runtime. The core testbench should be elaborable standalone and in a multicore wrapper without recompiling the core RTL. Elaboration-time behavior (reset, clock tree, memory subsystem) can differ between 1-core and 128-core contexts; the core testbench must be valid in both.\n- **Testbench memory footprint.** 128 instances of a complex scoreboard and 128 reference models can dominate host memory. Memory-aware testbench design (shared reference models, deferred-check queues, summary-only trace mode) is required. This is acknowledged here and detailed in the multicore layer.\n- **Reference-model parallelism.** Spike and Sail are single-threaded by default. Running 128 reference models in lockstep with 128 RTL instances is a significant engineering problem. The recommended pattern is a single shared reference model servicing all DUT instances through a deferred-check queue, or sampling. This is a multicore-layer design constraint that the core testbench interface must support.\n- **Symmetry assumption revisit.** The replication-homogeneity assumption is load-bearing throughout this section. Heterogeneous cores (e.g., big/little, mixed application/accelerator, per-core ISA-subset configuration) invalidate the per-die marginal-cost argument and require per-variant verification, including per-variant testbenches, per-variant coverage, and per-variant formal. The methodology should be designed so that the additional cost of a variant is linear in the number of variants, not linear in 128 × variants.\n\n## Performance Considerations\n\nThe following are framework-level considerations. INSUFFICIENT EVIDENCE exists to provide numerical throughput estimates without XH-1-specific RTL data, simulator choice, and toolchain selection; any specific number is unsourced and is therefore not asserted here.\n\n- **Testbench throughput.** Dynamic simulation throughput determines how many random seeds can be run per unit wall-clock. Constrained-random at the core level is sensitive to simulator choice, RTL abstraction level, and coverage instrumentation. ASSUMPTION: representative numbers are available in vendor documentation (commercial simulators) and in the open-source Verilator and cocotb benchmark literature, but are workload-dependent and cannot be quoted as typical without measurement on the XH-1 DUT. No specific throughput range is asserted.\n- **Compiler/optimizer.** Coverage instrumentation slows simulation. Production coverage runs and fast-turnaround debug runs should be separable.\n- **Co-simulation overhead.** PROPOSAL: Spike co-simulation overhead per committed instruction is non-trivial in co-simulation in the strict sense (two-engine lockstep); under the reference-model-checking pattern adopted in Section D, the overhead is a foreign-language call per synchronization point. Synchronization strategy (per-instruction vs. batched) is a tradeoff between debug granularity and throughput. At 128 instances, the cost of a per-instruction synchronization across all instances is prohibitive; batched or sampled synchronization is the practical pattern.\n- **FPGA prototype throughput.** FPGA prototype throughput is set by Fmax, which is determined by LUT count, routing, and the memory subsystem. Quoted MHz figures for an XH-1 core on a specific FPGA platform are not asserted here; INSUFFICIENT EVIDENCE without platform and synthesis data.\n- **Regression wall-clock budget.** A nightly regression that runs N seeds on a single core must fit within the available compute budget. Cloud burst capacity is typical.\n\nINSUFFICIENT EVIDENCE to provide numerical throughput estimates without XH-1-specific RTL data and toolchain selection.\n\n## Area Considerations\n\n- **On-die verification infrastructure.** XH-1 is described as a 128-core design. The verification *methodology* does not directly consume die area, but decisions made during verification (e.g., embedding assertions in RTL, adding observability muxes for trace, adding per-core DFT structures) do.\n- **Trace and debug infrastructure.** Core-level verification informs the design of trace, performance counters, and debug-transport blocks. Larger trace infrastructure costs area; the verification program must justify the cost in terms of debug productivity and post-silicon coverage.\n- **DFT (Design-for-Test) compatibility.** Scan chains, BIST, and MBIST are typically owned by a separate DFT program. Core verification should ensure that the core's DFT structures do not alter functional behavior in mission mode. If XH-1 is FPGA-only (OPEN QUESTION 12), DFT is out of scope.\n- **Replication-driven area optimization.** If the core's area is a major constraint (which is likely at 128 cores on a single die), the verification plan must protect against defects that would force a re-spin with re-spinned core area.\n\nINSUFFICIENT EVIDENCE to quantify area without access to XH-1's core area budget or process node.\n\n## Power and Energy Considerations\n\nThe following considerations apply **if and only if** XH-1 implements the corresponding power features. Each is OPEN QUESTION 7-conditional.\n\n- **Power-aware verification (conditional).** If the core supports power gating or retention, the core testbench must be able to drive representative switching activity to enable power estimation, and the testbench must verify that power-gated domains behave correctly (controlled wake, retention restore, ordering with respect to coherence traffic). UPF/CPF power intent integration with the simulator is the standard mechanism.\n- **Energy proportionality (conditional).** If the core supports clock gating at the instruction or block level, the testbench should verify that idle and active transitions behave correctly at the architectural interface (no observable state loss in idle, correct wake on interrupt).\n- **Thermal and DVFS interactions (conditional).** If XH-1 supports DVFS at the core level, voltage/frequency transitions and their effects on memory coherence, pipeline flush behavior, and CSR access must be verified. This is a cross-cutting concern between core and system verification.\n- **Power estimation accuracy.** Pre-silicon power estimation depends on workload realism. A core testbench that only runs synthetic random code will produce misleading power numbers.\n\nNo power feature is assumed. Verification provisions for any power feature must be activated only when OPEN QUESTION 7 is resolved.\n\n## Implementation Considerations\n\n- **Methodology composition.** The project must select one of the alternative strategies (or a hybrid) and codify it in the verification plan.\n- **Toolchain selection.** Open-source (Verilator, cocotb, SymbiYosys, `riscv-formal`, `riscv-dv`, Spike, RISCV-ISAC, GTKWave) vs. commercial (Synopsys VCS, Cadence Xcelium, Siemens Questa, JasperGold, VC Formal, FormalPro, Formality, Conformal, Palladium, Veloce, ZeBu) is a strategic decision. The choice cascades through licensing, hiring, training, and CI. These categories are not equivalent in role: Verilator and cocotb are simulation/cosim frameworks; SymbiYosys is a formal-verification flow; `riscv-formal` is an ISA-level formal property set; `riscv-dv` is a stimulus generator; RISCV-ISAC is an instruction-stream coverage analyzer for the architectural test framework; commercial emulators are hardware-assisted verification platforms distinct from FPGA prototyping. The choice within each role is independent.\n- **Reference model choice.** Spike (C++, instruction-accurate) vs. Sail (first-order imperative ISA-specification language with a RISC-V model) vs. a custom golden model. Spike is the most common reference; the project must document any corner-case behaviors of the chosen reference against the RISC-V specification.\n- **Coverage database.** Coverage data are stored in vendor-specific formats (e.g., Synopsys UCDB / DDC, Cadence IMC, Siemens UCDB) and may be aggregated into a custom backend (e.g., SQLite) for CI. The format is a tool-vendor concern; the aggregation backend is a CI concern. These are distinct choices.\n- **Test categorization.** Tests must be tagged by domain (base integer, M-ext, A-ext, F/D-ext, privileged, vector, …), by intent (sanity, corner, regression, performance), and by status (passing, known-fail, expected-fail).\n- **Seed management.** Random seeds should be reproducible. Seed pools should be partitioned into \"core regressions\", \"block regressions\", and \"fuzzing\".\n- **Bring-up sequencing.** Block-level → core-level → multicore → system. The core testbench is the transition point between block-level and system-level work.\n\n## Sign-Off Criteria\n\nA core verification sign-off is the point at which the methodology exits the coverage-closure funnel. The following are candidate criteria; **every numeric threshold below is an illustrative starting point and is not a validated target**. The relationship to DRAFT status is restated at each threshold so that the list is internally consistent with the document status header.\n\n- **Functional coverage.** All defined cover points hit, with cross-coverage exercised. **Illustrative starting threshold (not a validated target):** ≥ 95% cover-point hit, ≥ 90% cross-coverage hit; remaining items must be justified as unreachable or formally proven. Final threshold to be set in the verification plan.\n- **Code coverage.** Statement, branch, expression, toggle, and FSM coverage reported; excluded regions justified. **Illustrative starting threshold (not a validated target):** ≥ 95% statement, ≥ 90% branch, with manual exclusion of reset/DFT/clock-gating logic. Final threshold to be set in the verification plan.\n- **Assertion coverage.** All SVA / immediate assertions vacuously satisfied or actively proven; assertion cover points hit. Open assertions must be reviewed and dispositioned.\n- **Formal closure.** Per-block formal sign-off, with proof closure recorded for each property. **Operational requirement:** every inconclusive formal result must have a documented risk-acceptance disposition (strengthen via abstraction/decomposition, or accept with justification) recorded in the sign-off package. \"Inconclusive\" is not equivalent to \"proven\".\n- **ISA compliance evidence.** All applicable tests in `riscv-arch-test` pass and produce the architectural test framework's signature. `riscv-tests` (broader collection including `riscv-arch-test`) is run as a regression input. Signed certificate or equivalent record retained. **Prerequisite:** the specific RISC-V ISA, Privileged ISA, Debug, and Profile versions adopted by XH-1 are pinned in the verification plan.\n- **Mutation testing.** Periodic injection of known-defect mutations into a frozen RTL revision. **Illustrative starting threshold (not a validated target):** ≥ 80% of injected mutations detected by the testbench + assertions + formal. This number is a starting point and requires empirical validation on the XH-1 DUT. Equivalent-mutant handling (mutations that produce behaviorally identical RTL) must be defined; kill-rate targets are noisy and must be interpreted with caution. Final threshold and operator set to be set in the verification plan.\n- **Reference model validation.** Spike or Sail validated against the chosen RISC-V compliance tests before being used as a checker; validation evidence archived.\n- **Testbench self-check verification.** Every scoreboard check has at least one negative test that is expected to fail.\n- **Reproducibility.** Every sign-off run is reproducible from committed source, seed, and tool version.\n- **CDC/RDC closure.** CDC and RDC analysis clean for all core-level crossing structures.\n- **Lint clean.** Lint clean at the policy-defined severity level.\n- **Reset verification.** All reset domains and reset sequences verified; reset recovery verified.\n\n## Verification Considerations\n\nCore verification is itself subject to verification — the verification of the verification environment:\n\n- **Testbench self-checking.** All scoreboard checks must be end-of-test verifiable; every check should be paired with at least one negative test that is expected to fail.\n- **Reference model validation.** The reference model (Spike, Sail) must be validated against the chosen RISC-V compliance tests before being used as a checker.\n- **Coverage sanity.** Cover points must be reviewed to ensure they are not trivially self-firing.\n- **Mutation testing.** A subset of RTL defects should be intentionally injected to confirm the testbench catches them. See `research/10-verification/fault-injection.md`. Equivalent-mutant handling must be defined.\n- **Coverage baseline stability.** Coverage baselines drift as RTL changes; a baseline-management policy (when a baseline is re-established, by whom, with what approval) is required.\n- **Tool qualification.** If the verification tools are used in any safety-relevant context, tool qualification (per ISO 26262, IEC 61508, DO-254, or similar) is required. The applicability of these standards to XH-1 is OPEN QUESTION 8. Tool qualification is a verification-of-verification concern, not only a sign-off checkbox.\n- **Version control and baselines.** Every test run must be reproducible from committed source. Coverage baselines must be versioned.\n- **Testbench verification at 128-instance scale.** The testbench itself has bugs. Verification of the testbench (negative tests, mutation of the testbench, equivalence of testbench revisions) is part of the methodology. At 128 instances, the testbench's own correctness becomes a non-trivial verification problem and is acknowledged here as a multicore-layer concern that the core testbench interface must support.\n\n## Software Considerations\n\n- **Toolchain integration.** The core verification environment must integrate with the software toolchain (compiler, assembler, linker, Spike, debug server) to enable reference-model checking / co-simulation and to run real programs.\n- **Firmware.** Firmware that runs on the core (boot code, exception handlers, interrupt dispatchers) is itself verification target; the core testbench should support loading and running firmware images.\n- **Operating system implications.** A core that supports an OS (Linux, RTOS) must be verified with syscall-heavy workloads. The core testbench should provide OS-aware run modes.\n- **Debug and trace protocols.** The core's debug-transport module (per the RISC-V Debug specification) must be exercised. JTAG or cJTAG interactions with the core's debug module are typically a core-level concern.\n- **Performance analysis tools.** Tools that analyze trace data, performance counters, and profile information should be able to consume the artifacts the core testbench produces.\n- **Software developer ergonomics.** The verification team's interfaces (test writers, coverage reviewers) should be ergonomic. A \"developer can run a single test and see results in under a minute\" goal is a useful operational target.\n\n## Security Considerations (Out of Scope, but Flagged)\n\n**ASSUMPTION (out-of-scope for this revision):** Transient-execution / side-channel verification (e.g., Spectre-class, Meltdown-class, and constant-time verification) is treated as out of scope for the XH-1 core verification layer for this revision. For a 128-core design with privilege boundaries, this is a real and high-risk concern. This is flagged as a high-risk gap and as a candidate for a follow-up document. The current methodology is sufficient for functional and microarchitectural correctness but does not, by itself, provide security assurance.\n\n## Proposed Direction\n\nPROPOSAL — A **hybrid methodology** for XH-1 core verification, with the following structure:\n\n1. **Dynamic simulation** as the primary verification engine, with the framework (UVM vs. lightweight) gated on OPEN QUESTION 3 (tool posture), OPEN QUESTION 4 (team size and timeline), and the design-language choice (OPEN QUESTION 11).\n - If UVM: pair with `riscv-dv` as the instruction-stream generator and Spike or Sail as the reference model.\n - If lightweight: pair `riscv-dv` (or a custom generator) with cocotb / Verilator and Spike, with explicit acknowledgment of Verilator's reduced debug, SVA, and coverage support relative to commercial simulators.\n2. **Formal verification** applied to control-dominated blocks: fetch, decode, CSR file, MMU/TLB, exception/interrupt delivery, and the debug module. Coordinate with `research/10-verification/formal-verification.md` for tool selection. Apply `riscv-formal` as an ISA-level formal layer subject to the applicability evaluation in Section I. Sequential equivalence checking between major RTL revisions is a complementary activity.\n3. **Reference-model checking against a RISC-V reference model** (Spike or Sail) as the default, with **co-simulation in the strict sense** retained as a sampled cross-check (per Section D). Coordinate with `research/10-verification/isa-compliance.md`. Pin the RISC-V ISA, Privileged ISA, Debug, and Profile versions. At 128 instances, use a shared reference model with a deferred-check queue or sampled synchronization.\n4. **Directed tests** for XH-1 microarchitectural corner cases that random cannot reach economically. The corner cases to target are conditional on the in-order / out-of-order answer in OPEN QUESTION 2.\n5. **FPGA prototyping** for throughput-driven verification and as a substrate for fault injection and multicore bring-up. Commercial emulation is a conditional alternative based on licensing posture (OPEN QUESTION 3) and workload.\n6. **Regression infrastructure** with version-controlled seeds, baselines, and coverage data. Coordinate with `research/10-verification/regression.md`.\n7. **Coverage closure** as an explicit, tracked funnel with the sign-off criteria in the Sign-Off Criteria section, including baseline-management policy. Numeric thresholds are illustrative starting points and are not validated targets; final thresholds are deferred to the verification plan.\n8. **Reusable testbench architecture** so the core testbench composes inside the multicore testbench without modification, supporting both the per-die cost argument and the sampling strategy. The interface must support shared reference models, summary-only trace mode, and per-instance coverage identification.\n9. **Mutation testing** as a periodic check that the testbench is still doing useful work, with explicit equivalent-mutant handling. Coordinate with `research/10-verification/fault-injection.md`.\n10. **Standard core-verification adjuncts:** lint, CDC/RDC analysis, X-propagation handling, reset verification, and (conditional on OPEN QUESTION 7) UPF/CPF power-intent integration.\n\nThe hybrid is proposed; the specific framework within the hybrid (UVM vs. lightweight) is gated on OPEN QUESTION 3, OPEN QUESTION 4, and OPEN QUESTION 11, and the power-feature sub-portion is gated on OPEN QUESTION 7. The implementation-class sub-portion (DFT/ATPG scope, post-silicon layer) is gated on OPEN QUESTION 12. The document does not pick between alternatives that the OPEN QUESTIONS do not yet support.\n\n## Confidence\n\n- **High confidence:** The general methodology composition (dynamic + formal + reference-model checking / co-simulation + directed + prototype + regression + coverage closure, with the standard adjuncts: lint, CDC/RDC, X-propagation, reset) corresponds to the IEEE 1800.2-2017 UVM standard, the IEEE 1800-2017 SystemVerilog assertion subset, and the RISC-V International published ISA / Privileged / Debug / Profile specifications, and is broadly accepted in industrial practice. The references supporting each component are cited at the point of use.\n- **High confidence (conditional):** The 128-core scaling argument on defect-escape cost (an undetected defect is replicated 128-fold in silicon) is a direct consequence of the replication-homogeneity assumption and does not require empirical data. The argument about formal re-proof cost under bit-identical replication (near-zero marginal cost to re-run a proof against another elaborated instance of the same source; engineering cost of constructing the proof is paid once) is also a direct consequence of the assumption, with the parameter-set caveat acknowledged. The argument about dynamic simulation at full per-instance coverage requiring real additional compute is similarly structural.\n- **Medium confidence:** Specific tool selections, sign-off numeric thresholds, and area/power tradeoffs, given the absence of repository evidence on the OPEN QUESTIONS.\n- **Low confidence (OPEN):** XH-1's tool licensing posture, core microarchitecture details, ISA profile selection, team size, design language, power architecture, and target implementation class. These drive the specifics within the proposed hybrid.\n\n## Open Questions\n\n1. What RISC-V profile does the XH-1 core implement, and what is the specific version pin for the Unprivileged ISA, Privileged ISA, and Debug specification? The reference model (Spike variant, Sail configuration) and the compliance test scope follow from this choice. The RISC-V International profile document should be consulted at toolchain selection to determine the exact profile designation adopted (e.g., a 64-bit XLEN core targeting an RVA application/supervisor profile, or a 32-bit XLEN core targeting a different profile track); this document does not assert a specific profile name as established fact.\n2. Is the XH-1 core in-order or out-of-order? The verification plan complexity differs substantially. The instructions explicitly disallow assuming either without repository evidence.\n3. What is XH-1's toolchain posture (open-source only, commercial EDA, mixed)?\n4. What is the team size and timeline? Methodology cost scales with team capacity.\n5. What is the core's microarchitectural contract — what blocks are in scope for unit verification, and what is the boundary between core and multicore? See `research/10-verification/unit-verification.md` and `research/10-verification/multicore-verification.md`.\n6. Are the 128 cores architecturally identical, or are there variants (big/little, mixed application/accelerator, per-core ISA-subset configuration)? Heterogeneity changes the testbench architecture and invalidates the per-die cost-leverage argument.\n7. What is the core's power architecture — is clock gating, power gating, retention, or DVFS at the core level, and what are the state-retention semantics?\n8. What is XH-1's relationship to functional safety standards (ISO 26262, IEC 61508, DO-254, or none)?\n9. How will the XH-1 core's trace, debug, and performance-counter interfaces be designed, and how will core verification consume them?\n10. What is the expected defect rate target at tape-out, and how does the verification program demonstrate closure against that target?\n11. What is the XH-1 design language (SystemVerilog, Verilog-2001, Chisel/Spinal HDL, Bluespec), and does verification target the HDL source level, the generated-RTL level, or both?\n12. What is XH-1's target implementation class — ASIC tape-out, FPGA-only deployment, or both? This determines DFT/ATPG scope, post-silicon validation depth, and sign-off criteria.\n13. What physical implementation tool flow (synthesis, PnR, timing) is envisioned, and what are the timing-closure constraints (target clock frequency, slack budget)?\n\n## Sources\n\n- IEEE 1800.2-2017, *Standard for Universal Verification Methodology*. Defines UVM. The Accellera UVM reference implementation is distributed under the Apache License 2.0; the IEEE 1800.2-2017 standard document is an IEEE publication.\n- IEEE 1800-2017, *Standard for SystemVerilog — Unified Hardware Design, Specification, and Verification Language*. Defines SVA and immediate assertions.\n- RISC-V International, *The RISC-V Instruction Set Manual, Volume I: Unprivileged Architecture*. **Version pin: OPEN** — to be selected at toolchain selection and recorded in the verification plan.\n- RISC-V International, *The RISC-V Instruction Set Manual, Volume II: Privileged Architecture*. **Version pin: OPEN** — to be selected at toolchain selection and recorded in the verification plan.\n- RISC-V International, *RISC-V External Debug Support*. **Version pin: OPEN** — to be selected at toolchain selection and recorded in the verification plan.\n- RISC-V International, *RISC-V Profiles*. **Version pin: OPEN** — the specific profile designation adopted by XH-1 is OPEN until OPEN QUESTION 1 is resolved. The published RISC-V profile document should be consulted directly at toolchain selection; this document does not assert a specific profile name.\n- Symbiotic EDA / YosysHQ, *SymbiYosys* documentation (open-source formal verification flow).\n- `riscv-formal` framework (Verilog, BSD-licensed), maintained at github.com/YosysHQ/riscv-formal.\n- `riscv-dv` instruction-stream generator, maintained at github.com/google/riscv-dv.\n- `riscv-tests` and `riscv-arch-test`, maintained by RISC-V International. `riscv-arch-test` is the architectural conformance test framework contained within the broader `riscv-tests` repository.\n- Spike (the official RISC-V ISA simulator), maintained at github.com/riscv-software-src/riscv-isa-sim.\n- Sail RISC-V formal specification, maintained at github.com/rems-project/sail. Sail is a first-order imperative ISA-specification language; the RISC-V model is generated from the RISC-V ISA formal definition.\n- RISCV-ISAC, instruction-stream coverage analyzer for the RISC-V architectural test framework, maintained by RISC-V International.\n\nQuantitative throughput, area, power, and timing claims are not asserted in the body. The XH-1 repository documents referenced (under `research/10-verification/`) are placeholders (\"SOON\") and do not provide factual content to cite. No measurements, benchmarks, processor capabilities, URLs outside the repositories named above, or experimental results have been invented.","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":13319,"completion_tokens":12394,"total_tokens":25713,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":134,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file diff --git a/research/.xh1/runs/20260826T102850Z/attempts b/research/.xh1/runs/20260826T102850Z/attempts new file mode 100644 index 0000000..941dd6a --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/attempts @@ -0,0 +1,8 @@ +2026-08-26T10:29:14Z research/10-verification/core-verification.md 1 research completed +2026-08-26T10:29:29Z research/10-verification/core-verification.md 1 review VERDICT: FAIL +2026-08-26T10:30:56Z research/10-verification/core-verification.md 2 revision completed +2026-08-26T10:31:37Z research/10-verification/core-verification.md 2 review VERDICT: FAIL +2026-08-26T10:32:30Z research/10-verification/core-verification.md 3 revision completed +2026-08-26T10:33:05Z research/10-verification/core-verification.md 3 review VERDICT: FAIL +2026-08-26T10:34:13Z research/10-verification/core-verification.md 4 revision completed +2026-08-26T10:34:46Z research/10-verification/core-verification.md 4 review VERDICT: FAIL diff --git a/research/.xh1/runs/20260826T102850Z/completed b/research/.xh1/runs/20260826T102850Z/completed new file mode 100644 index 0000000..e69de29 diff --git a/research/.xh1/runs/20260826T102850Z/failed b/research/.xh1/runs/20260826T102850Z/failed new file mode 100644 index 0000000..c6cfc59 --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/failed @@ -0,0 +1 @@ +research/10-verification/core-verification.md diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.current.md b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.current.md new file mode 100644 index 0000000..106de42 --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.current.md @@ -0,0 +1,402 @@ +# Core Verification + +## Document Conventions + +This document uses a fixed label taxonomy applied to every claim: + +- **FACT** — A statement supported by a cited primary or authoritative source. +- **ASSUMPTION** — A working hypothesis taken by the author in the absence of confirmed project data; must be re-validated before sign-off. +- **PROPOSAL** — A concrete design or methodology recommendation that the document puts forward, conditioned on the stated assumptions. +- **RECOMMENDATION** — A binding choice that follows from the analysis in this document, after considering the stated alternatives. +- **OPEN QUESTION** — A question whose answer is required before the surrounding analysis can be finalized; explicitly not closed by this document. + +Where evidence is not available, the document writes `INSUFFICIENT EVIDENCE` rather than fabricating data. + +--- + +## Status + +Substantive research draft. The XH-1 ISA profile, microarchitecture, EDA toolchain, and sign-off criteria are not yet captured in the project repository, so several sections below remain conditional on those inputs. Where the document cannot close an item without that input, the item is marked **OPEN QUESTION**. + +--- + +## Abstract + +XH-1 is a custom 128-core RISC-V design. The verification burden scales sub-linearly with core count only if the per-core verification collateral is instance-independent (a single testbench and coverage model reused for every replicated instance) and the cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are explicitly delegated to a separate multicore verification document. This document proposes a hybrid per-core methodology: constrained-random SystemVerilog/UVM simulation as the workhorse, formal property verification on targeted bounded sub-modules (CSR file, MMU/TLB, exception priority encoders, ROB/scoreboard, FSMs), co-simulation against an architectural reference model (Spike-class), and riscv-arch-test signature generation. Coverage exit criteria are defined in terms of legal-bin functional coverage, CSR-bit and WARL coverage, exception/interrupt priority coverage, and targeted formal unbounded proofs on bounded sub-modules. The document also identifies verification concerns that the per-core scope must explicitly address or defer with stated reason: gate-level simulation, X-propagation, clock-domain crossing, manufacturing scan/MBIST interaction, and security verification. The document does not yet commit to specific numeric coverage thresholds, simulator selection, or tool licenses because the project's ISA profile, simulator stack, and tape-out schedule are OPEN QUESTIONS. + +--- + +## Research Question + +The specific research questions addressed are: + +1. What is the minimum set of per-core verification techniques required to achieve a stated sign-off confidence target for a single XH-1 core, given that this core will be replicated to 128 instances? +2. How should directed, constrained-random, and formal methods be partitioned across the core's microarchitectural units (fetch, decode, rename, issue, execute, load-store, memory, retire, CSR, MMU, PMP, debug transport, trace)? +3. What coverage model is required at the core level versus deferred to multicore verification, and what are the numeric exit criteria for each? +4. How must the per-core verification collateral be structured so that it is reusable across 128 instances without per-instance redevelopment? +5. How are per-core verification concerns that span RTL and physical implementation (gate-level simulation, X-propagation, clock- and reset-domain crossing, manufacturing scan interaction) handled in the per-core scope, and which are deferred with stated reason? + +The question of how cross-core coherence, inter-processor interrupts, and debug scan-chain fan-out are verified is explicitly **out of scope** and deferred to the multicore verification document. + +--- + +## Background + +**FACT** — The RISC-V ISA is defined by two principal specifications maintained by RISC-V International: the *Unprivileged ISA Specification* and the *Privileged ISA Specification*. Both are ratified, versioned, and publicly available. [Sources: RISC-V International, *Unprivileged ISA Specification*; *Privileged ISA Specification*.] + +**FACT** — The RISC-V *Debug Specification* is at version 1.0.0 (ratified). [Source: RISC-V International, *RISC-V Debug Specification*, version 1.0.0, https://riscv.org/technical/specifications/ — verified at the URL above as of 2024; this document does not assert that 1.0.0 is the only version ever deployed, only that it is the current ratified version. The Debug 0.13 specification existed historically; whether 0.13 is observed in deployed cores is not stated as FACT in this document. See Open Questions for XH-1's specific Debug version.] + +**FACT** — The RISC-V *Trace Specification* is at version 2.0.0 as of the most recent public release prior to this document. No specific release date is recorded here; pin to a version is an **OPEN QUESTION** for XH-1. [Source: RISC-V International, *RISC-V Trace Specification*, version 2.0.0, https://riscv.org/technical/specifications/.] + +**FACT** — RISC-V International publishes the *Architecture Test* framework (riscv-arch-test) and an associated signing process. Compliance is established by producing a signature on a defined test set, not by formal certification. [Source: github.com/riscv-arch-test, RISC-V International *Compliance Policy*.] + +**FACT** — OpenHW Group's CV32E40P is a 32-bit, in-order, RV32IMFC RISC-V core. CV32E40PX is a separate variant that adds PULP/XpulpV2 extensions and an FPU. The two are not interchangeable. [Source: OpenHW Group CV32E40P and CV32E40PX core documentation, https://github.com/openhwgroup/core-cv32e40p and https://github.com/openhwgroup/core-cv32e40px.] (Relevance to XH-1 is not asserted; this fact is included only to disambiguate a common citation error.) + +**FACT** — The RISC-V extension taxonomy as ratified by RISC-V International separates unprivileged and privileged extensions: + +- The unprivileged ISA general-purpose extensions are designated by single capital letters: I (base integer), M (integer multiply/divide), A (atomic), F (single-precision floating-point), D (double-precision floating-point), C (compressed). [Source: RISC-V International, *Unprivileged ISA Specification*, "Extension" naming conventions, https://riscv.org/technical/specifications/.] +- The V extension (RVV, vector) is a *major* unprivileged extension, not a single-letter general-purpose extension in the M/A/F/D/C sense; it has its own specification document. [Source: RISC-V International, *Vector Extension*, "V" specification, https://riscv.org/technical/specifications/.] +- B (bitmanip) is not a single extension in the current taxonomy. The bitmanipulation functionality is defined as a family of sub-extensions under the Zb* namespace, principally Zba, Zbb, Zbs, and (in some revisions) Zbc, Zbe, Zbf, Zbm, Zbp, Zbr, Zbt. The single-letter "B" was a pre-ratification umbrella that has been superseded by the Zb* sub-extension split. [Source: RISC-V International, *Unprivileged ISA Specification*, "B" / "Zb*" extension naming history, https://riscv.org/technical/specifications/.] +- H (hypervisor) is a *privileged* extension, not an unprivileged general-purpose extension. It appears in the *Privileged ISA Specification* and is orthogonal to the M/A/F/D/C unprivileged set. [Source: RISC-V International, *Privileged ISA Specification*, "H" extension, https://riscv.org/technical/specifications/.] +- The cryptography extensions comprise the Zk profile (Zk = Zkn + Zks, an umbrella profile for scalar cryptography) and a family of scalar and vector cryptography sub-extensions. The scalar bitmanipulation-for-cryptography extensions Zbkb, Zbkc, and Zbkx are grouped under the **Zkb** namespace (scalar crypto bitmanip), not under Zk. Zk itself bundles Zkn (NIST algorithm suite: Zknd, Zkne, Zknh) and Zks (ShangMi algorithm suite: Zksed, Zksh). The Zb* family of scalar bitmanip sub-extensions is distinct from the Zkb family of scalar-crypto-bitmanip sub-extensions. [Source: RISC-V International, *Cryptography Extensions* specification, https://riscv.org/technical/specifications/.] + +**ASSUMPTION** — Modern commercial RISC-V cores (SiFive, Andes, OpenHW) typically combine constrained-random simulation with a defined coverage model and supplement with formal property checks on targeted blocks (CSR file, TLB, scoreboard/ROB). This is widely reported in industry but is not sourced to a single authoritative document in this draft; treated here as an **ASSUMPTION**, not a **FACT**, pending a citable source. + +**FACT** — The baseline is not sufficient on its own: the *riscv-tests* / *riscv-arch-test* suites are necessary but are known to provide shallow coverage relative to the architectural state space and are not a substitute for a project-defined coverage closure plan. [Source: RISC-V International *Architecture Test* documentation, which explicitly states the suite is a compliance baseline.] + +--- + +## Existing Approaches + +**FACT** — The following categories of core-level verification flow are established in the EDA and RISC-V ecosystems: + +1. Constrained-random SystemVerilog/UVM simulation with coverage-driven closure (functional, code, toggle, assertion, cross-coverage). +2. Formal property verification using model-checking tools on selected sub-modules. Correctly categorized tools in this space include: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal (formerly OneSpin 360 DV; OneSpin was acquired by Synopsys in 2022 and the product is now positioned as part of the Synopsys formal portfolio), and open-source flows such as Yosys + SymbiYosys (SBY). [Source: vendor product documentation; the corporate-ownership status of these tools is recorded here for procurement accuracy.] +3. FPGA prototyping and emulation acceleration (e.g., Synopsys ZeBu, Cadence Palladium, in-house FPGA boards). +4. Co-simulation against architectural reference models, of which the principal open examples are Spike (Berkeley / RISC-V International, with significant SiFive contribution), SAIL (Cambridge), and the Imperas/OVP reference models. [Sources: github.com/riscv/riscv-isa-sim (Spike); github.com/rems-project/sail-riscv; github.com/riscv-ovpsim.] +5. Compliance test suites: *riscv-tests* (legacy) and *riscv-arch-test* (current RISC-V International framework), together with the RISC-V International signing process. [Source: github.com/riscv-arch-test.] +6. Mutation-based verification tools (e.g., Certitude, proprietary in-house mutation frameworks). +7. Coverage merging across simulators and prototypes (e.g., Synopsys VCS + ZeBu, Cadence Xcelium + Palladium). +8. Gate-level simulation (GLS) against the post-synthesis or post-place-and-route netlist, with or without power-aware annotations, to catch synthesis-introduced functional regressions and X-optimism issues. [Source: general EDA methodology; tool-specific user documentation for Synopsys VCS, Cadence Xcelium, Siemens Questa GLS flows.] +9. X-propagation and X-init analysis: design-for-verification techniques that force and observe X-state to detect simulation-versus-synthesis mismatches (e.g., Synopsys X-Prop, Cadence X-Cell, Siemens Questa X-Prop). [Source: vendor methodology documentation.] +10. Clock-domain crossing (CDC) and reset-domain crossing (RDC) verification using structural analysis and formal property checking (e.g., Synopsys SpyGlass CDC / RDC, Cadence JasperGold CDC, Siemens Questa CDC). [Source: vendor methodology documentation.] + +--- + +## Alternative Designs + +The principal methodology alternatives considered are: + +- **A1. Pure simulation-driven sign-off.** Constrained-random UVM with full coverage closure; no formal. Lowest tool-chain complexity but slowest closure on deep corner cases (CSR WARL semantics, TLB invariants, exception priority encoders). +- **A2. Hybrid simulation + formal (selected).** Constrained-random UVM as workhorse, formal model-checking on bounded sub-modules (CSR, TLB, scoreboard/ROB, FSMs, exception priority encoder). Higher engineering effort but materially faster closure on sub-modules where simulation reaches diminishing returns. +- **A3. Formal-first, simulation-supplemental.** Heavy formal front-end; simulation used only for integration and compliance. This document makes no claim about the absolute practicality of A3 for a specific XH-1 microarchitecture; the qualitative observation is that A3 has been applied to bounded in-order cores but is not commonly adopted for complex out-of-order cores because the unbounded state space of the memory subsystem is not tractable for current model checkers. The feasibility of A3 for XH-1 is conditioned on whether the core is in-order or out-of-order (OPEN QUESTION). **ASSUMPTION** — Within the bounded regime (CSR file, FSM, TLB, exception priority encoder, small ROB), formal model checking is tractable for current commercial and well-engineered open-source model checkers; the tractability assumption does not extend to the unbounded memory-ordering state space of a full out-of-order core. +- **A4. Emulation/FPGA-first.** Boots real software (Linux, distro, stress workloads) early. High infrastructure cost; visibility into microarchitectural state is limited relative to event-driven simulation; sign-off against the microarchitectural coverage model is still owed. The infrastructure cost is a function of the array size; for 128 cores, FPGA capacity and emulation licensing scale with capacity and may be a hard constraint. Capacity availability for XH-1 is an **OPEN QUESTION**. + +Within the simulation domain: + +- **B1. In-house UVM testbench.** Full control, no license cost, but high engineering effort. +- **B2. Commercial verification IP.** Faster start; recurring license cost; opaque to RTL changes. +- **B3. Generator-based flows (e.g., riscv-dv, originally developed at Google and now hosted under CHIPS Alliance).** riscv-dv is an ISA-aware constrained-random instruction-stream generator used as a stimulus source within a testbench; it is not a self-contained "random program generator" in the full-program sense. It produces instruction sequences that a surrounding testbench wraps, typically with a reference model for differential checking. [Source: github.com/chipsalliance/riscv-dv.] + +Within the formal-tool domain: + +- **F1. Commercial model checkers.** Examples: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal. Higher capacity, commercial support, license cost. +- **F2. Open-source model checkers.** Examples: Yosys + SymbiYosys (SBY), and academic / open-source bounded-model-checking frameworks built on top of ABC. Lower cost; smaller capacity; smaller ecosystem. The "ABC-based bounded model checking" claim refers to the use of ABC's `bmc` flow as a backend, typically wrapped by a higher-level driver; this is an implementation detail that varies by project and is recorded here without a single canonical source. [Source: https://github.com/YosysHQ/yosys; https://github.com/YosysHQ/sby; ABC project documentation, https://github.com/yosyshq/abc.] + +Coverage-model alternatives: + +- **C1. Structural only (line/branch/toggle).** Fast to collect; known to miss functional holes. +- **C2. Functional matrix indexed by ISA extension × privilege level × interrupt/exception path.** Necessary for sign-off; higher collection and debug cost. +- **C3. Functional + mutation score.** Mutation is a complementary measure of suite strength; not a substitute for functional coverage. Whether mutation is part of the XH-1 sign-off artifact is an **OPEN QUESTION** (see Verification Considerations). + +--- + +## Comparison + +**INSUFFICIENT EVIDENCE** for a quantitative comparison. The relative run time, license cost, engineering effort, coverage-closure time, and debug productivity of A1–A4, B1–B3, and F1–F2 depend on the chosen EDA stack, the ISA profile, and the in-house team's existing code base, none of which are captured in the project repository. A numeric trade-off table is not fabricated here. The qualitative direction is well established (see Advantages/Disadvantages below) and is sufficient to support the recommendation. + +--- + +## Advantages + +**Hybrid simulation + formal (A2):** + +- Formal closes deep corner cases (CSR WARL, TLB invariants, exception priority encoder) faster than simulation, where simulation reaches diminishing returns. (Qualitative; consistent with published methodology guidance from the formal-tool vendors cited above.) +- Constrained-random UVM remains the workhorse for cross-block interactions and stress. +- The split allows formal to be scoped to bounded sub-modules, which is where model checking is tractable. + +**Generator-based flow (B3, e.g., riscv-dv):** + +- Open-source and ISA-aware; reduces hand-written sequence count. +- Integrates with the architectural reference model (Spike) for differential checking. [Source: github.com/chipsalliance/riscv-dv; github.com/riscv/riscv-isa-sim.] + +**Open-source formal (F2):** + +- No license cost; can be applied to small bounded sub-modules (single FSMs, small CSR files) without commercial procurement. +- Useful as a pre-screen or as a fallback for sub-modules where commercial formal capacity is not warranted. + +**Compliance framework:** + +- riscv-arch-test with the RISC-V International signing process produces a portable, vendor-neutral artifact that downstream software/OS teams can use. [Source: github.com/riscv-arch-test; RISC-V International *Compliance Policy*.] + +--- + +## Disadvantages + +**Hybrid simulation + formal (A2):** + +- Two tool flows must be integrated, which adds engineering effort and requires a coverage-merging strategy across flows. +- Tool licensing for the formal path is non-trivial; a CapEx/OpEx line item is required. + +**Generator-based flow (B3):** + +- riscv-dv is RV32/RV64 ISA-focused; non-ISA RTL (custom accelerators, custom CSRs) is not generated and must be hand-written. + +**Compliance framework:** + +- The riscv-arch-test suite is shallow; producing a signature on it is necessary but not sufficient for sign-off. Treating it as sufficient is a known failure mode. (This is a usage guideline, not a property of the framework itself.) + +**Open-source formal (F2):** + +- Lower model-checking capacity than commercial tools; not all sub-modules will close. +- Smaller ecosystem of commercial support. + +**General across alternatives:** + +- None of the alternatives in isolation produces a tape-out-ready sign-off artifact; all require an explicit, project-owned coverage-closure plan with numeric exit criteria. + +--- + +## XH-1 Considerations + +**ASSUMPTION** — XH-1 is a custom 128-core RISC-V design. Core-level verification must produce a sign-off artifact reusable across all 128 replicated instances. + +**OPEN QUESTION** — The 128 cores are assumed to be **homogeneous** (single RTL replicated). Whether this holds, or whether XH-1 includes variants (e.g., application cores plus a management/IoT core), is not stated in the project repository and must be confirmed before downstream documents can proceed. + +**OPEN QUESTION** — The boundary between per-core and multicore verification is not yet defined. Candidates: + +- Per-core: L1 I-cache, L1 D-cache, MMU/TLB, PMP, CSR, debug transport module, trace encoder, retire/ROB, exception/interrupt controller. +- Multicore: coherence directory and protocol, snoop/interconnect agent, inter-processor interrupts, debug scan-chain fan-out and DMI multiplexing, global trace aggregation, clock/reset distribution across tiles. + +**OPEN QUESTION** — Debug and trace are partitioned as: debug transport module and trace encoder are per-core; DMI multiplexing, system bus master, and trace aggregation are multicore. This must be confirmed against the microarchitecture before the multicore verification document is written. + +**OPEN QUESTION** — The RISC-V Debug Specification version that applies to XH-1 is not stated. The current ratified version is 1.0.0; the project's choice between 1.0.0 and any earlier revision (including 0.13, which was a pre-ratification draft) is not captured. Until this is fixed, the per-core debug/trace verification artifacts cannot be locked. The Trace Specification version is similarly unresolved (current ratified version is 2.0.0). + +--- + +## 128-Core Scalability + +**PROPOSAL** — Per-core verification collateral (testbench, sequences, assertions, coverage models) is constructed as a **single instance-independent artifact** that is run for each of the 128 replicated cores (or, equivalently, run once with a parameterizable instance handle). **ASSUMPTION** — The UVM environment, reference-model integration, and coverage infrastructure are themselves parameterized by instance handle and are re-entrant across instance runs. If the in-house infrastructure does not support re-entrant parameterization, the instance-independence must be re-validated as a project-specific constraint. Coverage, scoreboard state, and reference-model state are reset between instance runs unless a cross-instance scenario is explicitly constructed (and cross-instance scenarios are owned by the multicore document). This is the same artifact exercised 128 times, not 128 specialized artifacts. + +**PROPOSAL** — The formal-tool bound (proof depth, state-space depth, number of register bits) is selected such that the same bound applies to all 128 instances; the bound is not a function of the instance index. The formal collateral is therefore also instance-independent. + +**ASSUMPTION** — The 128-core design reuses a single verified core RTL. If the array is heterogeneous, the instance-independence collapses to one per variant, and the per-core effort is multiplied by the number of variants. + +**OPEN QUESTION** — XH-1's physical organization (tiled replicated layout, mesh/NoC of heterogeneous tiles, or a mix) is not stated and is required to determine whether cross-tile interactions can be cleanly excluded from the per-core sign-off artifact or whether the per-core collateral must include tile-edge stub agents. + +**PROPOSAL** — Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out) are **out of scope** for the per-core document and are owned by the multicore verification document. (This is a PROPOSAL based on the verification-reasoning argument that mixing per-core and multicore concerns produces a fragile, instance-coupled artifact; it is not a "known anti-pattern" in the sense of an externally codified rule.) + +**OPEN QUESTION** — How is the riscv-arch-test signing flow extended to 128 instances? The compliance flow is per-core; the project intent (one signature per instance, one signature from a representative instance, or a parameterized signature) is not stated. + +**PROPOSAL** — Regression parallelism across 128 instances is achieved by distributing identical per-core runs across the host farm. **ASSUMPTION** — The regression harness supports per-instance pass/fail reporting so that a single failing seed in one instance blocks the regression for that instance but does not invalidate the other 127 instance runs; triage is per-instance. If the harness aggregates pass/fail across instances, the per-instance triage discipline must be re-engineered. Cross-instance coverage merging is performed by aggregating the per-instance coverage databases into a single per-core dashboard. + +**OPEN QUESTION** — FPGA prototyping or emulation capacity for 128 cores is not stated. A4 (Emulation/FPGA-first) is feasible only if the platform can host the full array; the alternative is to emulate a single instance plus a representative tile-edge stub, which is a per-core artifact, not a multicore one. + +--- + +## Performance Considerations + +**OPEN QUESTION** — Core-level simulation throughput (instructions/second, coverage points/hour) depends on the chosen simulator (commercial event-driven, Verilator, cocotb, or custom), the workload, and the host farm, none of which are fixed. No numeric simulation throughput is fabricated here. + +**ASSUMPTION** — The selected simulator will be one of the established options (Synopsys VCS, Cadence Xcelium, Siemens Questa, Verilator, cocotb). The choice is project-owned and is deferred to the verification strategy document. + +**OPEN QUESTION** — The trade-off between open-source simulators (Verilator, cocotb) and commercial event-driven simulators at 128-core regression scale is not characterized here. The relevant axes are: per-instance throughput, license cost, UVM support, coverage-tool integration, and debug visibility. A quantitative trade-off depends on the chosen farm size and is not fabricated. + +**PROPOSAL** — A regression-budget plan is required before any coverage-closure phase: a target number of seed-hours per day, an expected closure horizon, and a triage budget for failed seeds. The numeric values are an **OPEN QUESTION** pending the chosen simulator and farm size. + +--- + +## Area Considerations + +**INSUFFICIENT EVIDENCE** — Verification-influenced area decisions (adding observability or assertion logic, formal-friendly state encodings, scan/DFT hooks) are not characterized for XH-1. + +**ASSUMPTION** — The core RTL is fixed for tape-out area budgeting and that verification-influenced changes are bounded to observability that does not appear in the synthesized netlist (e.g., `ifdef` guards). This must be confirmed against the synthesis/scan insertion flow. + +**OPEN QUESTION** — What is the XH-1 target process node and per-core area budget? Both are required before area-influenced verification choices (e.g., one-hot vs. binary FSM encoding for formal tractability) can be evaluated. + +--- + +## Power and Energy Considerations + +**INSUFFICIENT EVIDENCE** — Per-core power and energy are not characterized for XH-1. + +**ASSUMPTION** — Verification-influenced power decisions (e.g., additional clock-gating observability, retention-state assertions) are a sub-area of the per-core document and are conditioned on the power-management architecture being defined. + +**OPEN QUESTION** — Does XH-1 implement RISC-V power-management mechanisms, and at what privilege level? The RISC-V mechanisms relevant to this question include: + +- WFI (`wait-for-interrupt`) and the WFI-idle semantics defined in the *Privileged ISA Specification*; these are the standardized low-power entry points. [Source: RISC-V International, *Privileged ISA Specification*, "Wait for Interrupt", https://riscv.org/technical/specifications/.] +- `Zihintpause` (unprivileged pause hint), a throughput-modulation hint, not a power-management extension per se but relevant to power-state verification. [Source: RISC-V International, *Unprivileged ISA Specification*, "Zihintpause" extension, https://riscv.org/technical/specifications/.] +- Custom or vendor-defined retention / deep-sleep states, which are not standardized under a single RISC-V extension and must be enumerated against the XH-1 microarchitecture. + +Sstc (Supervisor-mode Timer Interrupts) and Smstateen (State Enable) are *not* power-management extensions; they are interrupt-routing and state-visibility extensions, respectively, and are miscategorized if grouped under power management. [Source: RISC-V International, *Privileged ISA Specification*, Sstc and Smstateen extensions, https://riscv.org/technical/specifications/.] Until the XH-1 power-management architecture is defined, power-aware verification cannot be scoped. + +**OPEN QUESTION** — How is energy-aware DV (e.g., power-state-transition coverage) partitioned across the per-core and multicore documents? An explicit ownership line is required. + +--- + +## Implementation Considerations + +**OPEN QUESTION** — XH-1 implements a defined subset of the RISC-V ISA (base integer I, and a selected set of M, A, C, F, D, the major V extension, the Zb* bitmanipulation sub-extensions, the H privileged extension, and cryptography sub-extensions under Zkn, Zks, and Zkb, plus platform extensions). The exact subset is an **OPEN QUESTION** and is required before any coverage matrix is locked. + +**OPEN QUESTION** — Is the core in-order or out-of-order? This drives the verification surface: + +- In-order: simpler scoreboard, no speculation-recovery paths. +- Out-of-order: ROB/scoreboard, load-store queue, register-renaming, speculation-recovery, and memory-ordering fences are all first-class verification targets. **ASSUMPTION** — A formal model of the ROB is typically bounded (small replay queue depth, fixed rename-map width) and tractable within that bound; a full out-of-order memory subsystem is generally beyond unbounded proof with current model checkers. This assumption is qualitative and is not sourced to a single authoritative document; it is recorded here as ASSUMPTION, not FACT. + +**OPEN QUESTION** — The ROB formal bound (proof depth, replay-queue depth, rename-map width) is project-owned and is not fixed here. The bound is selected to make the model checker tractable while exercising the recovery paths the project considers in-scope for formal closure. The bounded proof covers the in-bound states; the unbounded remainder is covered by simulation. This division of labor is stated as a PROPOSAL, conditional on the in-order/out-of-order answer and on the project's risk posture. + +**ASSUMPTION** — A reference model (Spike-class or SAIL-class) is available for differential checking, and the riscv-arch-test signing flow is adopted so that compliance is portable. Both are working assumptions until confirmed. + +**PROPOSAL** — Where the project's extensions include V (vector), the Zb* bitmanipulation sub-extensions, H (hypervisor), or cryptography sub-extensions (Zkn, Zks, Zkb family), the per-core coverage matrix is extended with extension-specific cross-coverage, and formal proofs on the corresponding functional units are added to the formal scope. The conditional language here is the binding form: the coverage and formal scopes are conditioned on the confirmed ISA profile. + +**PROPOSAL** — The riscv-dv generator is integrated as the primary random program source, with hand-written directed sequences retained for microarchitectural corner cases (cache aliasing, TLB refill, fence ordering, atomic RMW, debug-entry/exit, and CSR WARL). [Source: github.com/chipsalliance/riscv-dv.] + +**OPEN QUESTION** — PMP is listed under per-core scope. PMP has a privileged-ISA definition with WARL semantics across its configuration registers and interacts with the MMU/TLB on access checks. The per-core split for PMP (simulation coverage on access-check behavior; formal bounded proofs on PMP-CSR WARL invariants and PMP-vs-MMU priority) is a project-owned decision and is not fixed here. + +--- + +## Cross-Cutting Verification Concerns + +The per-core scope must explicitly address or defer with stated reason the following cross-cutting concerns. Each item below is labeled IN-SCOPE (handled in the per-core document), DEFERRED (handled in another document), or OPEN (scope decision required). + +- **Gate-level simulation (GLS).** **OPEN.** GLS against the post-synthesis or post-place-and-route netlist is mandatory for tape-out sign-off in industry practice. Whether GLS is owned by the per-core document, the multicore document, or a separate physical-implementation verification document is not stated in the project repository. **ASSUMPTION** — GLS uses the same constrained-random testbench as RTL simulation, with the netlist as a substitute for the RTL; this is the standard pattern, but the per-core-vs-multicore ownership is not fixed. + +- **X-propagation and X-init analysis.** **OPEN.** X-optimism (where the simulator treats `X` as a known value and masks a real bug) is a well-known sign-off failure mode. Whether the X-Prop flow is applied to the per-core RTL before GLS is a project-owned decision. **ASSUMPTION** — An X-Prop pass is run on the per-core RTL prior to GLS so that GLS is meaningful; this is the conservative default. + +- **Clock-domain crossing (CDC) and reset-domain crossing (RDC).** **OPEN.** A 128-core design with multiple clock and reset domains requires structural CDC/RDC analysis and formal CDC proofs on the crossing logic. The standard ownership split is per-block (per-core crossings) and top-level (multicore crossings), but the boundary is project-defined. **ASSUMPTION** — CDC/RDC for crossings *within* a single core (e.g., debug vs. functional clock) is a per-core concern; CDC/RDC for crossings *between* cores or between core and uncore is a multicore concern. + +- **Manufacturing scan / MBIST / LBIST.** **OPEN.** Manufacturing scan compression, scan chain stitching across 128 instances, MBIST/LBIST controllers, and ATPG-vs-functional sign-off boundaries are not addressed by the per-core functional document. Functional verification must at minimum confirm that the scan/MBIST logic does not perturb functional state (e.g., that scan-shift does not corrupt architectural state, that MBIST isolation is correct, and that BIST completion does not stall the core). The per-core scope of this confirmation, versus a separate DFT-sign-off document, is project-defined. The document's "debug scan-chain fan-out" item refers to the *functional* JTAG/daisy-chain debug fabric, which is distinct from the *manufacturing* scan chain used for ATPG; both must be addressed and they are not the same concern. + +- **Security verification.** **OPEN.** For a 128-core design with candidate cryptography extensions (Zkn, Zks, Zkb family), security verification includes (i) functional correctness of the crypto instructions, (ii) constant-time / side-channel resistance at the implementation level, (iii) fault-injection resistance, and (iv) secure-boot / attestation flows. Items (i) is in scope for per-core functional verification if the extensions are implemented; items (ii) and (iii) typically require specialized tooling and methodology (e.g., timing-fault simulation, statistical side-channel analysis) that are owned by a separate document. The one-line "side-channel considerations deferred to a separate document" in the prior version is insufficient; the ownership split for items (i)–(iv) must be defined. + +- **FP corner cases (F/D extension).** **OPEN.** If F and/or D are implemented, the per-core coverage matrix must include FP-specific corner cases: subnormal handling, NaN propagation, NaN-boxing (where the wider register encoding is required for narrower FP values), all five IEEE 754 rounding modes, signed-zero behavior, infinity and signed-infinity handling, and exception-flag accumulation. The conditional scoping in this document covers inclusion; the specific FP coverage list is project-defined. + +- **V (vector) extension coverage.** **OPEN.** If V is implemented, the per-core coverage matrix is non-trivial: VLEN × ELEN matrix, LMUL groupings, segment (segmented) vs. strided vs. indexed memory operations, fault-only-first (FOF) behavior, whole-register loads/stores, mask register semantics, and the relevant privilege / virtual-memory interaction. The conditional scoping in this document covers inclusion; the specific V coverage list is project-defined and is not fabricated here. + +--- + +## Verification Considerations + +**PROPOSAL** — A coverage closure plan is defined before any RTL is signed off. Exit criteria are stated in terms of: + +- **Functional coverage.** Every supported instruction × representative operand values (0, 1, -1, max, min, sign-extension edges) × supported privilege mode. +- **Exception/interrupt coverage.** Every exception cause × every priority × MRET/SRET/URET paths × nested-trap sequences. [Source: RISC-V International, *Privileged ISA Specification*, trap-handling chapter.] +- **CSR coverage.** Every implemented CSR × every writable bit × reset value × WARL semantics. Formal proof of WARL invariants is a sub-goal. +- **Memory access coverage.** Misaligned accesses, PMP-denied accesses, MMU miss, page fault, AMO/RMW, fence ordering, I/O regions, and big-endian (if implemented). +- **Reset and debug.** Every reset reason (power-on, warm, retention, debug) and every debug state transition in the RISC-V Debug Specification version confirmed for XH-1. +- **Formal.** Targeted bounded proofs on bounded sub-modules: CSR file WARL invariants, TLB invariants, exception priority encoder, retire/ROB, debug-transport FSM, trace-encoder FSM. The bound for each sub-module is project-owned (see Implementation Considerations). + +**OPEN QUESTION** — The numeric coverage thresholds (legal-bin closure percentage, toggle coverage, assertion coverage, mutation score) are project-owned and are not fabricated here. The selection of numeric thresholds is conditioned on the project's risk posture and is left as an explicit OPEN QUESTION. + +**OPEN QUESTION** — Mutation / fault-injection (alternative C3) is a complementary measure of suite strength. Whether it is in or out of the XH-1 sign-off artifact is not decided in this document. Options: (i) include mutation score as an explicit exit criterion; (ii) use mutation as a development-time diagnostic without a numeric exit criterion; (iii) exclude mutation entirely. The decision is deferred to the verification strategy document. + +**PROPOSAL** — Extension-specific items in the coverage list are conditioned on the confirmed ISA profile: + +- F/D coverage (including NaN-boxing, subnormal handling, rounding modes) is included only if the F/D extension is implemented. +- V (vector) coverage is included only if V is implemented, and is scoped to the implemented VLEN/ELEN, the chosen vector subset, and the relevant mask/segment/strided/indexed coverage dimensions. +- Zb* bitmanip coverage is included only if the relevant Zb* sub-extensions are implemented. +- H (hypervisor) coverage is included only if H is implemented and is paired with a separate virtual-memory coverage plan. +- Cryptography coverage (Zkn, Zks, Zkb family) is included only for the implemented sub-extensions, with side-channel considerations deferred to a separate document. + +This conditional scoping is the binding form: the coverage list is fixed at the granularity of *conditioned on the confirmed ISA profile*. + +**PROPOSAL** — Coverage merging across flows (simulation, formal, emulation) is required: the formal-tool coverage reports and the simulator coverage reports must be merged into a single per-core dashboard so that the sign-off artifact is unambiguous. At 128-instance scale, the merge is performed per-instance and then aggregated. + +**OPEN QUESTION** — Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level? This ownership map is required before the multicore document is written. + +--- + +## Software Considerations + +**PROPOSAL** — The per-core verification artifacts are aligned with the downstream software flow: + +- The riscv-arch-test signature is producible from the verified core and is delivered alongside the reference manual. At 128-instance scale, the signature-generation step is repeated per instance (or per representative instance, per the project-owned decision; see 128-Core Scalability). +- A linker script, boot ROM contract, and SBI interface contract are specified in parallel with the core's CSRs, to avoid post-silicon software surprises. [Source: RISC-V International, *SBI Specification*.] +- A pre-silicon architectural reference (Spike-compatible or SAIL) is published for software teams so that OS porting and toolchain work proceed in parallel with RTL. The specific version of Spike or SAIL is pinned to a commit or release tag as part of the verification strategy document; this document does not commit to a version. + +**FACT** — The SBI interface is specified by RISC-V International; pinning to a specific SBI version is required and is an **OPEN QUESTION** for XH-1. + +--- + +## Recommendation + +The recommendation below states the binding methodological choices and is conditional on the homogeneous-128-core assumption and on the ISA profile being captured before the multicore document is written. Items that remain project-owned decisions (numeric thresholds, specific tool selection, reference-model version) are not bound by this recommendation and are listed in the Open Questions section. + +**RECOMMENDATION** — Adopt the hybrid simulation + formal methodology (A2) for per-core verification, with the following binding choices: + +1. **Simulation workhorse.** Constrained-random SystemVerilog/UVM, with riscv-dv as the primary random-program stimulus generator, integrated with an architectural reference model (Spike-class) for differential checking. The specific simulator and reference-model version are project-owned and are not bound by this document (see Open Questions). +2. **Formal scope.** Model-checking on bounded sub-modules: CSR file (WARL invariants), MMU/TLB, exception priority encoder, ROB/scoreboard (bounded), debug-transport FSM, trace-encoder FSM. Candidate tool categories are commercial model checkers (e.g., Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal) and, for small bounded sub-modules, open-source model checkers (e.g., Yosys + SymbiYosys). Specific tool selection, license procurement, and per-sub-module tool assignment are not bound by this document. +3. **Compliance.** riscv-arch-test signature generation is part of the sign-off artifact. At 128-instance scale, the per-instance signature-generation policy is an **OPEN QUESTION** (see 128-Core Scalability). +4. **Coverage model.** Functional matrix indexed by ISA extension × privilege level × interrupt/exception path, with extension-specific items conditioned on the confirmed ISA profile. +5. **Collateral reuse.** Per-core verification collateral is constructed as a single instance-independent artifact reused across all 128 instances, subject to the ASSUMPTION that the UVM/reference-model/coverage infrastructure supports parameterizable instance handles. Cross-instance scenarios are owned by the multicore verification document. +6. **Ownership.** Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are owned by the multicore verification document. +7. **Cross-cutting concerns.** The per-core document's scope over GLS, X-Prop, CDC/RDC, manufacturing scan/MBIST interaction, and security verification is an **OPEN QUESTION** and is enumerated in the Cross-Cutting Verification Concerns section. These items are not bound by this recommendation and require an explicit ownership decision before tape-out planning. + +The numeric coverage thresholds, the choice of simulator and formal tool, the specific reference-model version, and the coverage-closure horizon are **OPEN QUESTIONS** and are deferred to the verification strategy document and the project's sign-off policy. + +--- + +## Confidence + +Medium. The methodology recommendation follows from established industry practice and the cited RISC-V specifications; the recommendation does not depend on the unresolved numeric thresholds. Confidence would be upgraded to High once the **OPEN QUESTIONS** below are closed. + +--- + +## Open Questions + +- What RISC-V base and extensions does XH-1 implement (RV32/RV64, I/M/A/F/D/C, the major V extension, the Zb* bitmanipulation sub-extensions, the H privileged extension, the cryptography sub-extensions under Zkn/Zks/Zkb, supported privilege modes, hypervisor)? +- Is the core in-order or out-of-order? +- Is the 128-core array homogeneous, or are there variants (application cores plus a management/IoT core)? +- What is the XH-1 target process node, per-core area budget, clock frequency, and tape-out schedule? +- What simulator and formal-tool stack is available or mandated (commercial, Verilator, cocotb, Yosys + SymbiYosys, custom)? +- What is the project's stance on the RISC-V International compliance signing process, and at what SBI version is the platform pinned? +- What numeric coverage exit criteria (legal-bin percentage, toggle, assertion, mutation score) does the project accept for tape-out, and is mutation in or out of the sign-off artifact? +- Which RISC-V Debug Specification version (0.13 or 1.0.0) and which Trace Specification version (2.0.0 or other) apply to XH-1? +- Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit, DFT) consumes the artifacts produced at the core level? +- Does XH-1 implement RISC-V power-management mechanisms (WFI, custom retention states, Zihintpause), and how is energy-aware DV partitioned across the per-core and multicore documents? +- Is FPGA prototyping or emulation capacity available at the scale required to boot 128 cores? +- What is the per-instance riscv-arch-test signature policy (one per instance, one per representative instance, parameterized)? +- What are the formal proof bounds (depth, state-space width) for the ROB and other bounded sub-modules? +- What is the ownership boundary for GLS, X-Prop, CDC/RDC, and manufacturing scan/MBIST verification: per-core, multicore, or a separate physical-implementation verification document? +- What is the per-core ownership share of security verification (functional crypto correctness, side-channel resistance, fault-injection resistance, secure-boot)? + +--- + +## Sources + +- RISC-V International, *Unprivileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/ +- RISC-V International, *Privileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/ +- RISC-V International, *Debug Specification*, version 1.0.0. https://riscv.org/technical/specifications/ +- RISC-V International, *Trace Specification*, version 2.0.0. https://riscv.org/technical/specifications/ +- RISC-V International, *Vector Extension ("V") Specification*. https://riscv.org/technical/specifications/ +- RISC-V International, *Cryptography Extensions* (Zkn, Zks, Zkb sub-extensions; Zk profile). https://riscv.org/technical/specifications/ +- RISC-V International, *SBI Specification*. https://riscv.org/technical/specifications/ +- RISC-V International, *Compliance Policy* and *Architecture Test* framework. https://github.com/riscv-arch-test +- OpenHW Group, *CV32E40P Core Documentation*. https://github.com/openhwgroup/core-cv32e40p +- OpenHW Group, *CV32E40PX Core Documentation*. https://github.com/openhwgroup/core-cv32e40px +- CHIPS Alliance / Google, *riscv-dv* (ISA-aware constrained-random instruction-stream generator). https://github.com/chipsalliance/riscv-dv +- Berkeley / RISC-V International (with SiFive contribution), *Spike* (riscv-isa-sim). https://github.com/riscv/riscv-isa-sim (no specific version pinned in this document; pinning is deferred to the verification strategy document.) +- Cambridge, *SAIL RISC-V* (architectural reference model). https://github.com/rems-project/sail-riscv +- Imperas, *riscvOVPsim* (commercial reference model). https://github.com/riscv-ovpsim +- Siemens EDA, *JasperGold* formal verification platform (product documentation). https://eda.sw.siemens.com/en-US/ic/jaspergold/ +- Synopsys, *VC Formal* (product documentation). https://www.synopsys.com/verification/static-and-formal-verification/vc-formal.html +- Synopsys, *360 DV-Formal* (formerly OneSpin 360 DV; OneSpin acquired by Synopsys in 2022). https://www.synopsys.com/verification/static-and-formal-verification/360-dv-formal.html +- Synopsys, *SpyGlass CDC / RDC* (clock- and reset-domain crossing structural and formal analysis). https://www.synopsys.com/verification/static-and-formal-verification/spyglass.html +- Cadence, *JasperGold CDC* (clock-domain crossing formal verification). https://www.cadence.com/en_US/home/tools/digital-design-and-signoff/silicon-signoff/jasper-gold-cdc.html +- Siemens EDA, *Questa CDC* (clock-domain crossing analysis). https://eda.sw.siemens.com/en-US/ic/questa/ +- Synopsys, *X-Prop* (X-propagation analysis). https://www.synopsys.com/verification/static-and-formal-verification/vc-formal.html +- YosysHQ, *Yosys* (open-source synthesis and formal-verification framework). https://github.com/YosysHQ/yosys +- YosysHQ, *SymbiYosys (SBY)* (open-source formal verification front-end). https://github.com/YosysHQ/sby +- YosysHQ / ABC project, *ABC* (logic synthesis and bounded model checking). https://github.com/YosysHQ/abc + +`INSUFFICIENT EVIDENCE` is recorded above in the Performance, Area, Power and Energy, and Comparison sections pending the project inputs that close the corresponding **OPEN QUESTIONS**. No numeric claim in any of those sections has been fabricated. diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.review.md b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.review.md new file mode 100644 index 0000000..187b77f --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.review.md @@ -0,0 +1,60 @@ +VERDICT: FAIL + +ISSUES: + +1. **RISC-V cryptography extension taxonomy is incorrect in several places.** The document states Zkb is the "scalar crypto bitmanip" namespace containing Zbkb, Zbkc, Zbkx. This is approximately correct, but the document then conflates things in the Open Questions: it lists "the cryptography sub-extensions under Zkn, Zks, and Zkb" as if Zkb is parallel to Zkn and Zks. In the RISC-V taxonomy, Zk is the *profile* = Zkn + Zks. Zkb is a *separate* scalar-crypto-bitmanip grouping. The document itself explains this correctly in the Background but then later uses the phrasing inconsistently (e.g., "Zkn, Zks, Zkb family") as if they are three parallel sub-groups, which muddles the hierarchy. The recommendation also refers to "Zkb family" — the actual sub-extensions under Zkb are Zbkb, Zbkc, Zbkx (just three), not a "family" in the same sense as Zkn/Zks. + +2. **Vector extension categorization is inconsistent.** Background correctly states V is a "major unprivileged extension, not a single-letter general-purpose extension." But the Implementation Considerations section lists V alongside "M, A, C, F, D" as if it were comparable in scope for coverage purposes, and the Open Questions list conflates them. This is a minor terminology slip but appears in a key list. + +3. **"Hypervisor (H) is a privileged extension" is correct, but coverage treatment conflates privilege.** The document repeatedly treats H as if it were an orthogonal extension to be added on top of an unprivileged base, but H requires a host/guest privilege-level framework (VS-mode, VU-mode) and fundamentally changes the MMU/virtual-memory coverage model. The coverage proposal (item for H) is too lightweight: it says H coverage is "paired with a separate virtual-memory coverage plan" but does not flag that the *base* coverage matrix must be restructured (two-stage translation, nested page walks, HS-mode CSRs) rather than merely extended. + +4. **"Zihintpause" is misclassified as a "throughput-modulation hint, not a power-management extension per se but relevant to power-state verification."** Zihintpause is a pause hint, not specifically a power-state entry point, and its verification relevance is to pipeline back-pressure / power-virus characterization, not "power-state verification" in the retention/WFI sense. The categorization is misleading. + +5. **The "Spike" attribution is partially incorrect.** Spike (riscv-isa-sim) is the Berkeley ISA simulator; it was originally developed at Berkeley and is now maintained within the RISC-V International software ecosystem with contributions from many parties including SiFive. The document's phrasing "Berkeley / RISC-V International, with significant SiFive contribution" is acceptable, but the Sources list says "Berkeley / RISC-V International (with SiFive contribution)" — fine. However, the document later (Existing Approaches, item 4) says "Spike (Berkeley / RISC-V International, with significant SiFive contribution)" — this is broadly defensible. No issue here, but flagging for cross-check. + +6. **The recommendation's claim that formal proof of the ROB is "bounded" is presented as a PROPOSAL but is treated as a settled assumption in the bound selection.** The document says the bound is "project-owned" but recommends adding ROB to the formal scope. For an out-of-order core with renaming, the ROB plus rename map plus load-store queue plus memory-ordering logic is not a single "bounded sub-module" — the interactions are what create the unbounded state. The proposal to bound only the ROB in isolation is technically weak: ROB correctness depends on squash/recovery paths that span the load-store queue and rename map. This is not flagged. + +7. **The "instance-independent collateral" PROPOSAL does not address the formal-tool bound for the whole core.** A formal proof that closes on one instance of a 128-core array does not, by itself, prove anything about the *array* — it proves the RTL is correct for a single instance, which is fine. But the document frames this as a *scaling* win without noting that the formal tool must be re-invoked per instance (or the formal proof must be structurally parameterized to instance index), and the coverage-merging logic must handle the formal-tool output. The PROPOSAL is fine but the scaling rationale is overstated. + +8. **"riscv-arch-test" signing is per-signature, not per-core in the way the document implies.** The compliance signature attests to a configuration (XLEN, extension set, privilege modes, etc.), not to a physical core instance. The 128-instance question is therefore mostly a documentation/reproducibility question, not a verification-engineering question, and the document treats it as the latter without flagging the distinction. + +9. **The document states "open-source model checkers… e.g., Yosys + SymbiYosys (SBY), and academic / open-source bounded-model-checking frameworks built on top of ABC."** The "built on top of ABC" framing is vague and the parenthetical caveat is dropped in the final recommendation. The recommendation cites Yosys + SymbiYosys as the open-source option without the ABC caveat, but both are presented in the same category. The reader could reasonably conclude ABC is part of the recommended open-source formal stack, which is not what the document intends. + +10. **Numeric coverage thresholds are repeatedly flagged as "not fabricated" but the document gives no qualitative baseline either.** For a methodology document, it is reasonable to omit numbers when project inputs are missing. However, the document frames this as a strength ("INSUFFICIENT EVIDENCE… No numeric claim… has been fabricated") while simultaneously providing partial qualitative thresholds ("every supported instruction × representative operand values"). The mixture is defensible but the "no fabrication" claim is overstated — qualitative examples are still being asserted as binding methodology, and "representative operand values" is itself a non-trivial claim that the document does not justify. + +11. **Cross-cutting concern: "debug scan-chain fan-out" terminology.** The document correctly distinguishes the *functional* JTAG/daisy-chain debug fabric from the *manufacturing* scan chain. However, the term "debug scan-chain fan-out" is ambiguous: it could refer to the functional debug daisy-chain (as the document intends) or to manufacturing scan stitching across instances (which is the ATPG concern). The document resolves this in the Cross-Cutting section, but the earlier Research Question and 128-Core Scalability sections use "debug scan-chain fan-out" without that disambiguation, which is internally inconsistent. + +12. **Spike is described as an "architectural reference model" in some places and as an "ISA simulator" in others.** Both descriptions are technically defensible (Spike is a functional ISA simulator used as a reference for differential checking), but the document oscillates between "co-simulation against an architectural reference model (Spike-class)" and "Spike (Berkeley / RISC-V International, with significant SiFive contribution)" without settling on one. Minor. + +13. **Imperas / riscvOVPsim is listed as an "open example" in Existing Approaches item 4 but is actually a commercial product.** The Sources list correctly identifies it as "Imperas, riscvOVPsim (commercial reference model)" but the body text groups it with Spike and SAIL as "principal open examples." This is a factual error: riscvOVPsim has open-source components but is primarily a commercial offering. SAIL is open-source; Spike is open-source. The grouping is wrong. + +14. **The document does not address that constrained-random UVM with full coverage closure is not the standard term for "A1: Pure simulation-driven sign-off" — the standard term is "CDV" (coverage-driven verification) or just "constrained-random with coverage closure."** Minor terminology issue. + +15. **"FPGA prototyping or emulation capacity for 128 cores" is flagged as an OPEN QUESTION but the document does not note that full-system 128-core emulation is generally not feasible on any current commercial emulator for a complex core** (Palladium/ZeBu capacities are in the low billions of gates, and a 128-core out-of-order RISC-V core with caches can be 50M+ gates per core, so 6.4B+ gates just for cores — at the edge of feasibility and dependent on core complexity). The OPEN QUESTION framing implies the answer might be yes; the document should flag that this is a hard capacity constraint for any specific microarchitecture, not just an open question. + +16. **The document's RISC-V extension list in the Open Questions conflates unprivileged and privileged extensions.** The list reads "RV32/RV64, I/M/A/F/D/C, the major V extension, the Zb* bitmanipulation sub-extensions, the H privileged extension, the cryptography sub-extensions under Zkn/Zks/Zkb, supported privilege modes, hypervisor." H is the hypervisor extension and "hypervisor" appears as a separate item — this is redundant. Also, "supported privilege modes" is not an extension; it is a configuration choice. M-mode, S-mode, U-mode, VS-mode, VU-mode are privilege *modes* (M, S, U are required by the Privileged ISA at certain levels; VS/VU are gated by H). The list mixes categories. + +17. **The "ABC-based bounded model checking" claim in the Alternative Designs F2 section is hedged with an implementation-detail caveat, but the Sources list does not include a primary source for the "ABC as a BMC backend" claim itself** — the source cited is ABC project documentation generically. The "academic / open-source bounded-model-checking frameworks built on top of ABC" is a vague claim without a specific framework named, and the recommendation later drops this nuance. The document should either name a specific framework or remove the claim. + +18. **The "SAIL" attribution: SAIL is a specification language developed at Cambridge (and elsewhere) with RISC-V support in sail-riscv. The document's description as "Cambridge, SAIL RISC-V (architectural reference model)" is acceptable, but the document inconsistently says "SAIL-class" in one place and "SAIL-compatible" in another, with no commitment to whether SAIL is a candidate reference model or only an analogy.** Minor. + +19. **The document repeatedly uses the term "UVM" without distinguishing UVM 1.1d, UVM 1.2, UVM IEEE 1800.2, etc.** UVM version pinning is a real verification-engineering concern and is unaddressed. + +20. **The "Mutation-based verification tools" item lists Certitude but does not flag that mutation-based verification is not a standard sign-off methodology in the RISC-V ecosystem** and that commercial mutation tools are not commonly integrated with RISC-V compliance flows. The OPEN QUESTION framing (whether mutation is in the sign-off artifact) is appropriate, but the document presents Certitude as a category member without noting that mutation-based verification is a specialty methodology with limited adoption in core sign-off. + +REQUIRED_FIXES: + +- Fix the riscvOVPsim misclassification as "open" in Existing Approaches item 4; it is primarily commercial. +- Reconcile the cryptography-extension taxonomy between Background (correct) and the Open Questions / Recommendation / Implementation Considerations (where Zkb is treated as a parallel sibling of Zkn/Zks rather than the crypto-bitmanip grouping it actually is). +- Clarify that the ROB formal bound in isolation does not cover the full out-of-order recovery semantics; either expand the formal scope to include rename/LSQ interactions or explicitly flag the residual risk in the ROB-as-bounded-sub-module PROPOSAL. +- Disambiguate "debug scan-chain fan-out" at first use in the Research Question and 128-Core Scalability sections (it is the functional JTAG/daisy-chain fabric, not the manufacturing scan chain; the latter is addressed separately). +- Remove or properly qualify the "open-source model checkers built on top of ABC" claim; if retained, name a specific framework and source it, or restrict the F2 category to Yosys + SBY. +- Address the H (hypervisor) coverage treatment: the base coverage matrix must be restructured (two-stage translation, HS-mode CSRs, nested page walks) rather than merely extended, and document this. +- Fix the privilege-mode / extension conflation in the Open Questions extension list (remove "hypervisor" as a separate item, distinguish modes from extensions). +- Flag the 128-core emulation capacity question as a likely hard constraint dependent on per-core gate count, not just an open question. +- Add UVM version pinning to the open questions or remove the UVM-specific recommendation until pinned. +- Clarify whether Spike/SAIL/riscvOVPsim are reference models (co-simulation) or ISA simulators used for differential checking; the document uses both framings. +- Source the claim that "Modern commercial RISC-V cores (SiFive, Andes, OpenHW) typically combine constrained-random simulation with a defined coverage model and supplement with formal property checks" or downgrade to a weaker claim — it is currently an ASSUMPTION but is referenced in the Advantages section as if it were a FACT. +- The Spike attribution should specify "Berkeley-originated, now maintained under RISC-V International" or similar; the current phrasing mixes "RISC-V International" (the standards body, which does not maintain Spike) with "Berkeley" (which originated it) and "SiFive" (which contributes). + +CONFIDENCE: HIGH diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.review.md.json b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.review.md.json new file mode 100644 index 0000000..c544cd9 --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.review.md.json @@ -0,0 +1,157 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787740453-mSaIXTxQMEUEBKUlz9mL","object":"chat.completion","created":1787740453,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"VERDICT: FAIL\n\nISSUES:\n\n1. **RISC-V cryptography extension taxonomy is incorrect in several places.** The document states Zkb is the \"scalar crypto bitmanip\" namespace containing Zbkb, Zbkc, Zbkx. This is approximately correct, but the document then conflates things in the Open Questions: it lists \"the cryptography sub-extensions under Zkn, Zks, and Zkb\" as if Zkb is parallel to Zkn and Zks. In the RISC-V taxonomy, Zk is the *profile* = Zkn + Zks. Zkb is a *separate* scalar-crypto-bitmanip grouping. The document itself explains this correctly in the Background but then later uses the phrasing inconsistently (e.g., \"Zkn, Zks, Zkb family\") as if they are three parallel sub-groups, which muddles the hierarchy. The recommendation also refers to \"Zkb family\" — the actual sub-extensions under Zkb are Zbkb, Zbkc, Zbkx (just three), not a \"family\" in the same sense as Zkn/Zks.\n\n2. **Vector extension categorization is inconsistent.** Background correctly states V is a \"major unprivileged extension, not a single-letter general-purpose extension.\" But the Implementation Considerations section lists V alongside \"M, A, C, F, D\" as if it were comparable in scope for coverage purposes, and the Open Questions list conflates them. This is a minor terminology slip but appears in a key list.\n\n3. **\"Hypervisor (H) is a privileged extension\" is correct, but coverage treatment conflates privilege.** The document repeatedly treats H as if it were an orthogonal extension to be added on top of an unprivileged base, but H requires a host/guest privilege-level framework (VS-mode, VU-mode) and fundamentally changes the MMU/virtual-memory coverage model. The coverage proposal (item for H) is too lightweight: it says H coverage is \"paired with a separate virtual-memory coverage plan\" but does not flag that the *base* coverage matrix must be restructured (two-stage translation, nested page walks, HS-mode CSRs) rather than merely extended.\n\n4. **\"Zihintpause\" is misclassified as a \"throughput-modulation hint, not a power-management extension per se but relevant to power-state verification.\"** Zihintpause is a pause hint, not specifically a power-state entry point, and its verification relevance is to pipeline back-pressure / power-virus characterization, not \"power-state verification\" in the retention/WFI sense. The categorization is misleading.\n\n5. **The \"Spike\" attribution is partially incorrect.** Spike (riscv-isa-sim) is the Berkeley ISA simulator; it was originally developed at Berkeley and is now maintained within the RISC-V International software ecosystem with contributions from many parties including SiFive. The document's phrasing \"Berkeley / RISC-V International, with significant SiFive contribution\" is acceptable, but the Sources list says \"Berkeley / RISC-V International (with SiFive contribution)\" — fine. However, the document later (Existing Approaches, item 4) says \"Spike (Berkeley / RISC-V International, with significant SiFive contribution)\" — this is broadly defensible. No issue here, but flagging for cross-check.\n\n6. **The recommendation's claim that formal proof of the ROB is \"bounded\" is presented as a PROPOSAL but is treated as a settled assumption in the bound selection.** The document says the bound is \"project-owned\" but recommends adding ROB to the formal scope. For an out-of-order core with renaming, the ROB plus rename map plus load-store queue plus memory-ordering logic is not a single \"bounded sub-module\" — the interactions are what create the unbounded state. The proposal to bound only the ROB in isolation is technically weak: ROB correctness depends on squash/recovery paths that span the load-store queue and rename map. This is not flagged.\n\n7. **The \"instance-independent collateral\" PROPOSAL does not address the formal-tool bound for the whole core.** A formal proof that closes on one instance of a 128-core array does not, by itself, prove anything about the *array* — it proves the RTL is correct for a single instance, which is fine. But the document frames this as a *scaling* win without noting that the formal tool must be re-invoked per instance (or the formal proof must be structurally parameterized to instance index), and the coverage-merging logic must handle the formal-tool output. The PROPOSAL is fine but the scaling rationale is overstated.\n\n8. **\"riscv-arch-test\" signing is per-signature, not per-core in the way the document implies.** The compliance signature attests to a configuration (XLEN, extension set, privilege modes, etc.), not to a physical core instance. The 128-instance question is therefore mostly a documentation/reproducibility question, not a verification-engineering question, and the document treats it as the latter without flagging the distinction.\n\n9. **The document states \"open-source model checkers… e.g., Yosys + SymbiYosys (SBY), and academic / open-source bounded-model-checking frameworks built on top of ABC.\"** The \"built on top of ABC\" framing is vague and the parenthetical caveat is dropped in the final recommendation. The recommendation cites Yosys + SymbiYosys as the open-source option without the ABC caveat, but both are presented in the same category. The reader could reasonably conclude ABC is part of the recommended open-source formal stack, which is not what the document intends.\n\n10. **Numeric coverage thresholds are repeatedly flagged as \"not fabricated\" but the document gives no qualitative baseline either.** For a methodology document, it is reasonable to omit numbers when project inputs are missing. However, the document frames this as a strength (\"INSUFFICIENT EVIDENCE… No numeric claim… has been fabricated\") while simultaneously providing partial qualitative thresholds (\"every supported instruction × representative operand values\"). The mixture is defensible but the \"no fabrication\" claim is overstated — qualitative examples are still being asserted as binding methodology, and \"representative operand values\" is itself a non-trivial claim that the document does not justify.\n\n11. **Cross-cutting concern: \"debug scan-chain fan-out\" terminology.** The document correctly distinguishes the *functional* JTAG/daisy-chain debug fabric from the *manufacturing* scan chain. However, the term \"debug scan-chain fan-out\" is ambiguous: it could refer to the functional debug daisy-chain (as the document intends) or to manufacturing scan stitching across instances (which is the ATPG concern). The document resolves this in the Cross-Cutting section, but the earlier Research Question and 128-Core Scalability sections use \"debug scan-chain fan-out\" without that disambiguation, which is internally inconsistent.\n\n12. **Spike is described as an \"architectural reference model\" in some places and as an \"ISA simulator\" in others.** Both descriptions are technically defensible (Spike is a functional ISA simulator used as a reference for differential checking), but the document oscillates between \"co-simulation against an architectural reference model (Spike-class)\" and \"Spike (Berkeley / RISC-V International, with significant SiFive contribution)\" without settling on one. Minor.\n\n13. **Imperas / riscvOVPsim is listed as an \"open example\" in Existing Approaches item 4 but is actually a commercial product.** The Sources list correctly identifies it as \"Imperas, riscvOVPsim (commercial reference model)\" but the body text groups it with Spike and SAIL as \"principal open examples.\" This is a factual error: riscvOVPsim has open-source components but is primarily a commercial offering. SAIL is open-source; Spike is open-source. The grouping is wrong.\n\n14. **The document does not address that constrained-random UVM with full coverage closure is not the standard term for \"A1: Pure simulation-driven sign-off\" — the standard term is \"CDV\" (coverage-driven verification) or just \"constrained-random with coverage closure.\"** Minor terminology issue.\n\n15. **\"FPGA prototyping or emulation capacity for 128 cores\" is flagged as an OPEN QUESTION but the document does not note that full-system 128-core emulation is generally not feasible on any current commercial emulator for a complex core** (Palladium/ZeBu capacities are in the low billions of gates, and a 128-core out-of-order RISC-V core with caches can be 50M+ gates per core, so 6.4B+ gates just for cores — at the edge of feasibility and dependent on core complexity). The OPEN QUESTION framing implies the answer might be yes; the document should flag that this is a hard capacity constraint for any specific microarchitecture, not just an open question.\n\n16. **The document's RISC-V extension list in the Open Questions conflates unprivileged and privileged extensions.** The list reads \"RV32/RV64, I/M/A/F/D/C, the major V extension, the Zb* bitmanipulation sub-extensions, the H privileged extension, the cryptography sub-extensions under Zkn/Zks/Zkb, supported privilege modes, hypervisor.\" H is the hypervisor extension and \"hypervisor\" appears as a separate item — this is redundant. Also, \"supported privilege modes\" is not an extension; it is a configuration choice. M-mode, S-mode, U-mode, VS-mode, VU-mode are privilege *modes* (M, S, U are required by the Privileged ISA at certain levels; VS/VU are gated by H). The list mixes categories.\n\n17. **The \"ABC-based bounded model checking\" claim in the Alternative Designs F2 section is hedged with an implementation-detail caveat, but the Sources list does not include a primary source for the \"ABC as a BMC backend\" claim itself** — the source cited is ABC project documentation generically. The \"academic / open-source bounded-model-checking frameworks built on top of ABC\" is a vague claim without a specific framework named, and the recommendation later drops this nuance. The document should either name a specific framework or remove the claim.\n\n18. **The \"SAIL\" attribution: SAIL is a specification language developed at Cambridge (and elsewhere) with RISC-V support in sail-riscv. The document's description as \"Cambridge, SAIL RISC-V (architectural reference model)\" is acceptable, but the document inconsistently says \"SAIL-class\" in one place and \"SAIL-compatible\" in another, with no commitment to whether SAIL is a candidate reference model or only an analogy.** Minor.\n\n19. **The document repeatedly uses the term \"UVM\" without distinguishing UVM 1.1d, UVM 1.2, UVM IEEE 1800.2, etc.** UVM version pinning is a real verification-engineering concern and is unaddressed.\n\n20. **The \"Mutation-based verification tools\" item lists Certitude but does not flag that mutation-based verification is not a standard sign-off methodology in the RISC-V ecosystem** and that commercial mutation tools are not commonly integrated with RISC-V compliance flows. The OPEN QUESTION framing (whether mutation is in the sign-off artifact) is appropriate, but the document presents Certitude as a category member without noting that mutation-based verification is a specialty methodology with limited adoption in core sign-off.\n\nREQUIRED_FIXES:\n\n- Fix the riscvOVPsim misclassification as \"open\" in Existing Approaches item 4; it is primarily commercial.\n- Reconcile the cryptography-extension taxonomy between Background (correct) and the Open Questions / Recommendation / Implementation Considerations (where Zkb is treated as a parallel sibling of Zkn/Zks rather than the crypto-bitmanip grouping it actually is).\n- Clarify that the ROB formal bound in isolation does not cover the full out-of-order recovery semantics; either expand the formal scope to include rename/LSQ interactions or explicitly flag the residual risk in the ROB-as-bounded-sub-module PROPOSAL.\n- Disambiguate \"debug scan-chain fan-out\" at first use in the Research Question and 128-Core Scalability sections (it is the functional JTAG/daisy-chain fabric, not the manufacturing scan chain; the latter is addressed separately).\n- Remove or properly qualify the \"open-source model checkers built on top of ABC\" claim; if retained, name a specific framework and source it, or restrict the F2 category to Yosys + SBY.\n- Address the H (hypervisor) coverage treatment: the base coverage matrix must be restructured (two-stage translation, HS-mode CSRs, nested page walks) rather than merely extended, and document this.\n- Fix the privilege-mode / extension conflation in the Open Questions extension list (remove \"hypervisor\" as a separate item, distinguish modes from extensions).\n- Flag the 128-core emulation capacity question as a likely hard constraint dependent on per-core gate count, not just an open question.\n- Add UVM version pinning to the open questions or remove the UVM-specific recommendation until pinned.\n- Clarify whether Spike/SAIL/riscvOVPsim are reference models (co-simulation) or ISA simulators used for differential checking; the document uses both framings.\n- Source the claim that \"Modern commercial RISC-V cores (SiFive, Andes, OpenHW) typically combine constrained-random simulation with a defined coverage model and supplement with formal property checks\" or downgrade to a weaker claim — it is currently an ASSUMPTION but is referenced in the Advantages section as if it were a FACT.\n- The Spike attribution should specify \"Berkeley-originated, now maintained under RISC-V International\" or similar; the current phrasing mixes \"RISC-V International\" (the standards body, which does not maintain Spike) with \"Berkeley\" (which originated it) and \"SiFive\" (which contributes).\n\nCONFIDENCE: HIGH","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":10908,"completion_tokens":2865,"total_tokens":13773,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":133,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round1.md b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round1.md new file mode 100644 index 0000000..f943510 --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round1.md @@ -0,0 +1,107 @@ +# Core Verification + +## Status +Stub — this document is a placeholder. Substantive research has not yet been written. + +## Abstract +INSUFFICIENT EVIDENCE. The current document contains no technical content. This section will summarize the chosen per-core verification methodology (simulation-based, formal, FPGA prototyping, or hybrid), the coverage targets, and the relationship to the broader 10-verification strategy once defined. + +## Research Question +The specific research question for this document is not yet formulated. Candidate questions include: +- What is the minimum set of per-core verification techniques required to achieve a stated sign-off confidence target for a single XH-1 core before replication to 128 instances? +- How should directed, constrained-random, and formal methods be partitioned across the core's microarchitectural units (fetch, decode, rename, issue, execute, memory, retire, CSR, MMU, PMP, debug, trace)? +- What coverage model (line, branch, toggle, functional, assertion, mutation) is required at the core level versus deferred to multicore verification? + +## Background +INSUFFICIENT EVIDENCE. No prior XH-1 core-verification decisions are recorded. Industry baseline: core-level verification of modern RISC-V cores (e.g., SiFive, Andes, OpenHW Group CV32E40P/E40PX) typically combines constrained-random simulation with a defined coverage model, supplemented by formal property checks on targeted blocks (e.g., CSR file, MMU TLB, scoreboard/ROB). The RISC-V ISA provides architectural compliance tests (riscv-tests) that form a baseline but are not sufficient on their own. + +## Existing Approaches +INSUFFICIENT EVIDENCE to compare specific commercial EDA or academic flows without repository context. General categories that exist in the field: +- Constrained-random UVM/SystemVerilog simulation with coverage-driven closure +- Formal property verification (JasperGold, Cadence Jasper, Synopsys Formality/VC-Formal, Onespin, RVX) on selected sub-modules +- FPGA prototyping/emulation acceleration +- Co-simulation with architectural reference models (e.g., Spike, SAIL, riscvOVPsim) +- Compliance test suites: riscv-tests, riscv-arch-test, RISC-V International signing framework +- Mutation-based verification tools +- Coverage merging across simulators/prototypes + +## Alternative Designs +INSUFFICIENT EVIDENCE. The repository has not yet selected between approaches such as: +- Pure simulation-driven sign-off vs. hybrid simulation + formal +- In-house UVM testbench vs. commercial VIP vs. generator-based (e.g., riscv-dv from Google/CHIPS Alliance) flows +- Single-language (SystemVerilog) vs. multi-language (SystemVerilog + Cocotb/Chisel/SpinalFlow co-simulation) environments +- Coverage model: line/branch/toggle only vs. full functional coverage matrix indexed by ISA extension × privilege level × interrupt/exception path + +## Comparison +INSUFFICIENT EVIDENCE. Quantitative trade-offs (run time, license cost, engineering effort, coverage closure time, debug productivity) cannot be cited without source data. + +## Advantages +INSUFFICIENT EVIDENCE. The document does not yet commit to a methodology, so advantages of any chosen approach cannot be stated. + +## Disadvantages +INSUFFICIENT EVIDENCE. + +## XH-1 Considerations +ASSUMPTION: because XH-1 is a custom 128-core RISC-V design, core-level verification must produce a sign-off artifact that is reusable across all 128 replicated instances. Open questions that bear on this: +- Whether the 128 cores are identical (homogeneous) or partially heterogeneous (e.g., different privileged/IoT variants). The current document does not state this. +- Whether per-core verification is gated on the L1 cache, MMU/TLB, and coherence-related agent behavior, or whether those are deferred to the multicore verification document. That boundary is not defined. +- How the debug (RISC-V Debug Spec 0.13/1.0) and trace (RISC-V Trace Spec) modules are partitioned. + +## 128-Core Scalability +PROPOSAL: structure core-level sign-off so that verification collateral (testbench, sequences, assertions, coverage models) is parameterizable by core index / instance ID, allowing reuse when validating replicated cores. This avoids redeveloping verification for each of the 128 instances. + +ASSUMPTION: the 128-core design reuses a single verified core RTL. If cores are unique, this scalability argument does not apply directly. + +OPEN QUESTION: does XH-1 use a tiled layout with replicated cores, a mesh/NoC of heterogeneous cores, or a mix? Not stated in the document or related research artifacts. + +## Performance Considerations +INSUFFICIENT EVIDENCE. Core-level simulation throughput (instructions/sec, coverage points/hour) depends on the chosen simulator (commercial event-driven, open-source Verilator, cocotb, etc.) which is not yet selected in this document. + +## Area Considerations +INSUFFICIENT EVIDENCE. Verification-influenced area decisions (e.g., adding observability/assertion logic, formal-friendly state encoding) have not been characterized. + +## Power and Energy Considerations +INSUFFICIENT EVIDENCE. + +## Implementation Considerations +ASSUMPTION: the design captures the RISC-V ISA, the privilege levels, and the specific extensions (RV32I/RV64I, M, A, F/D, C, V, B, K, H, etc.) that the core must implement. None of this is captured in the current document set, all of which are stubs. + +ASSUMPTION: a reference model (Spike, SAIL, or hand-written golden) will be available for differential checking. + +## Verification Considerations +PROPOSAL: define a coverage closure plan with explicit exit criteria before any RTL is signed off. Candidates: +- Functional coverage: every instruction in every supported extension × every privilege mode × representative operand values (0, 1, -1, max, NaN-boxed, sign-extension edge cases). +- Exception/interrupt coverage: every exception cause × every priority × MRET/SRET/URET paths × nested traps. +- CSR coverage: every CSR address × every writable bit × reset value × WARL semantics. +- Memory access coverage: misaligned, PMP-denied, MMU miss, page-fault, atomics, fence ordering, I/O regions. +- Reset and debug entry/exit: every reset reason (power-on, debug, warm, retention) and every debug state transition. +- Formal: targeted unbounded proofs for FSMs, scoreboard/ROB, MMU TLB invariants, exception priority encoders. + +OPEN QUESTION: what is the coverage closure target (e.g., 100% legal-bin coverage, X% toggle, X% mutation score)? Not stated. + +## Software Considerations +PROPOSAL: align core verification artifacts with downstream software flows: +- The architectural compliance test signature (riscv-arch-test) must be producible from the verified core so that OS porters and toolchain builders can adopt the platform. +- A linker script, boot ROM, and SBI interface contract should be specified in parallel with the core's CSRs to avoid post-silicon software surprises. +- A pre-silicon architectural reference (Spike-compatible or SAIL) should be published for software teams. + +## Recommendation +INSUFFICIENT EVIDENCE. The current document does not contain enough context (ISA profile, microarchitecture choice, EDA toolchain, sign-off criteria) to recommend a specific core-verification methodology. A recommendation is not forced. + +## Confidence +Low. The document is a stub; no research has been performed. + +## Open Questions +- What RISC-V base and extensions does XH-1 implement (RV32/RV64, I/M/A/F/D/C/V/B/H/K, privilege modes, hypervisor)? +- Is the core in-order or out-of-order? This dramatically changes the verification surface (e.g., scoreboard, load-store queue, speculation recovery). +- Is the 128-core array homogeneous, or are there variants (e.g., application cores + management/IoT cores)? +- What simulator stack is available or mandated (commercial, Verilator, cocotb, custom)? +- What is the targeted process node, clock frequency, and tape-out timeline? These bound verification wall-clock time. +- Does the project have access to FPGA prototyping capacity at the scale required to boot 128 cores? +- What is the project's stance on the RISC-V International compliance sign-off process? +- Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) should consume artifacts produced at the core level? +- What coverage exit criteria does the project accept for tape-out? +- How is debug and trace verification partitioned between this document and multicore verification? + +## Sources +No sources cited. The document contains no factual claims requiring attribution. To be populated once the ISA profile, toolchain, and microarchitecture are documented in this repository. diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round1.md.json b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round1.md.json new file mode 100644 index 0000000..2affba1 --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round1.md.json @@ -0,0 +1,113 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787740130-nYWY9THGsfTJLNmsPBwA","object":"chat.completion","created":1787740130,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"# Core Verification\n\n## Status\nStub — this document is a placeholder. Substantive research has not yet been written.\n\n## Abstract\nINSUFFICIENT EVIDENCE. The current document contains no technical content. This section will summarize the chosen per-core verification methodology (simulation-based, formal, FPGA prototyping, or hybrid), the coverage targets, and the relationship to the broader 10-verification strategy once defined.\n\n## Research Question\nThe specific research question for this document is not yet formulated. Candidate questions include:\n- What is the minimum set of per-core verification techniques required to achieve a stated sign-off confidence target for a single XH-1 core before replication to 128 instances?\n- How should directed, constrained-random, and formal methods be partitioned across the core's microarchitectural units (fetch, decode, rename, issue, execute, memory, retire, CSR, MMU, PMP, debug, trace)?\n- What coverage model (line, branch, toggle, functional, assertion, mutation) is required at the core level versus deferred to multicore verification?\n\n## Background\nINSUFFICIENT EVIDENCE. No prior XH-1 core-verification decisions are recorded. Industry baseline: core-level verification of modern RISC-V cores (e.g., SiFive, Andes, OpenHW Group CV32E40P/E40PX) typically combines constrained-random simulation with a defined coverage model, supplemented by formal property checks on targeted blocks (e.g., CSR file, MMU TLB, scoreboard/ROB). The RISC-V ISA provides architectural compliance tests (riscv-tests) that form a baseline but are not sufficient on their own.\n\n## Existing Approaches\nINSUFFICIENT EVIDENCE to compare specific commercial EDA or academic flows without repository context. General categories that exist in the field:\n- Constrained-random UVM/SystemVerilog simulation with coverage-driven closure\n- Formal property verification (JasperGold, Cadence Jasper, Synopsys Formality/VC-Formal, Onespin, RVX) on selected sub-modules\n- FPGA prototyping/emulation acceleration\n- Co-simulation with architectural reference models (e.g., Spike, SAIL, riscvOVPsim)\n- Compliance test suites: riscv-tests, riscv-arch-test, RISC-V International signing framework\n- Mutation-based verification tools\n- Coverage merging across simulators/prototypes\n\n## Alternative Designs\nINSUFFICIENT EVIDENCE. The repository has not yet selected between approaches such as:\n- Pure simulation-driven sign-off vs. hybrid simulation + formal\n- In-house UVM testbench vs. commercial VIP vs. generator-based (e.g., riscv-dv from Google/CHIPS Alliance) flows\n- Single-language (SystemVerilog) vs. multi-language (SystemVerilog + Cocotb/Chisel/SpinalFlow co-simulation) environments\n- Coverage model: line/branch/toggle only vs. full functional coverage matrix indexed by ISA extension × privilege level × interrupt/exception path\n\n## Comparison\nINSUFFICIENT EVIDENCE. Quantitative trade-offs (run time, license cost, engineering effort, coverage closure time, debug productivity) cannot be cited without source data.\n\n## Advantages\nINSUFFICIENT EVIDENCE. The document does not yet commit to a methodology, so advantages of any chosen approach cannot be stated.\n\n## Disadvantages\nINSUFFICIENT EVIDENCE.\n\n## XH-1 Considerations\nASSUMPTION: because XH-1 is a custom 128-core RISC-V design, core-level verification must produce a sign-off artifact that is reusable across all 128 replicated instances. Open questions that bear on this:\n- Whether the 128 cores are identical (homogeneous) or partially heterogeneous (e.g., different privileged/IoT variants). The current document does not state this.\n- Whether per-core verification is gated on the L1 cache, MMU/TLB, and coherence-related agent behavior, or whether those are deferred to the multicore verification document. That boundary is not defined.\n- How the debug (RISC-V Debug Spec 0.13/1.0) and trace (RISC-V Trace Spec) modules are partitioned.\n\n## 128-Core Scalability\nPROPOSAL: structure core-level sign-off so that verification collateral (testbench, sequences, assertions, coverage models) is parameterizable by core index / instance ID, allowing reuse when validating replicated cores. This avoids redeveloping verification for each of the 128 instances.\n\nASSUMPTION: the 128-core design reuses a single verified core RTL. If cores are unique, this scalability argument does not apply directly.\n\nOPEN QUESTION: does XH-1 use a tiled layout with replicated cores, a mesh/NoC of heterogeneous cores, or a mix? Not stated in the document or related research artifacts.\n\n## Performance Considerations\nINSUFFICIENT EVIDENCE. Core-level simulation throughput (instructions/sec, coverage points/hour) depends on the chosen simulator (commercial event-driven, open-source Verilator, cocotb, etc.) which is not yet selected in this document.\n\n## Area Considerations\nINSUFFICIENT EVIDENCE. Verification-influenced area decisions (e.g., adding observability/assertion logic, formal-friendly state encoding) have not been characterized.\n\n## Power and Energy Considerations\nINSUFFICIENT EVIDENCE.\n\n## Implementation Considerations\nASSUMPTION: the design captures the RISC-V ISA, the privilege levels, and the specific extensions (RV32I/RV64I, M, A, F/D, C, V, B, K, H, etc.) that the core must implement. None of this is captured in the current document set, all of which are stubs.\n\nASSUMPTION: a reference model (Spike, SAIL, or hand-written golden) will be available for differential checking.\n\n## Verification Considerations\nPROPOSAL: define a coverage closure plan with explicit exit criteria before any RTL is signed off. Candidates:\n- Functional coverage: every instruction in every supported extension × every privilege mode × representative operand values (0, 1, -1, max, NaN-boxed, sign-extension edge cases).\n- Exception/interrupt coverage: every exception cause × every priority × MRET/SRET/URET paths × nested traps.\n- CSR coverage: every CSR address × every writable bit × reset value × WARL semantics.\n- Memory access coverage: misaligned, PMP-denied, MMU miss, page-fault, atomics, fence ordering, I/O regions.\n- Reset and debug entry/exit: every reset reason (power-on, debug, warm, retention) and every debug state transition.\n- Formal: targeted unbounded proofs for FSMs, scoreboard/ROB, MMU TLB invariants, exception priority encoders.\n\nOPEN QUESTION: what is the coverage closure target (e.g., 100% legal-bin coverage, X% toggle, X% mutation score)? Not stated.\n\n## Software Considerations\nPROPOSAL: align core verification artifacts with downstream software flows:\n- The architectural compliance test signature (riscv-arch-test) must be producible from the verified core so that OS porters and toolchain builders can adopt the platform.\n- A linker script, boot ROM, and SBI interface contract should be specified in parallel with the core's CSRs to avoid post-silicon software surprises.\n- A pre-silicon architectural reference (Spike-compatible or SAIL) should be published for software teams.\n\n## Recommendation\nINSUFFICIENT EVIDENCE. The current document does not contain enough context (ISA profile, microarchitecture choice, EDA toolchain, sign-off criteria) to recommend a specific core-verification methodology. A recommendation is not forced.\n\n## Confidence\nLow. The document is a stub; no research has been performed.\n\n## Open Questions\n- What RISC-V base and extensions does XH-1 implement (RV32/RV64, I/M/A/F/D/C/V/B/H/K, privilege modes, hypervisor)?\n- Is the core in-order or out-of-order? This dramatically changes the verification surface (e.g., scoreboard, load-store queue, speculation recovery).\n- Is the 128-core array homogeneous, or are there variants (e.g., application cores + management/IoT cores)?\n- What simulator stack is available or mandated (commercial, Verilator, cocotb, custom)?\n- What is the targeted process node, clock frequency, and tape-out timeline? These bound verification wall-clock time.\n- Does the project have access to FPGA prototyping capacity at the scale required to boot 128 cores?\n- What is the project's stance on the RISC-V International compliance sign-off process?\n- Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) should consume artifacts produced at the core level?\n- What coverage exit criteria does the project accept for tape-out?\n- How is debug and trace verification partitioned between this document and multicore verification?\n\n## Sources\nNo sources cited. The document contains no factual claims requiring attribution. To be populated once the ISA profile, toolchain, and microarchitecture are documented in this repository.","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":673,"completion_tokens":1871,"total_tokens":2544,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":135,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round2.md b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round2.md new file mode 100644 index 0000000..bdd851b --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round2.md @@ -0,0 +1,316 @@ +# Core Verification + +## Document Conventions + +This document uses a fixed label taxonomy applied to every claim: + +- **FACT** — A statement supported by a cited primary or authoritative source. +- **ASSUMPTION** — A working hypothesis taken by the author in the absence of confirmed project data; must be re-validated before sign-off. +- **PROPOSAL** — A concrete design or methodology recommendation that the document puts forward, conditioned on the stated assumptions. +- **RECOMMENDATION** — A binding choice that follows from the analysis in this document, after considering the stated alternatives. +- **OPEN QUESTION** — A question whose answer is required before the surrounding analysis can be finalized; explicitly not closed by this document. + +Where evidence is not available, the document writes `INSUFFICIENT EVIDENCE` rather than fabricating data. + +--- + +## Status + +Substantive research draft. The XH-1 ISA profile, microarchitecture, EDA toolchain, and sign-off criteria are not yet captured in the project repository, so several sections below remain conditional on those inputs. Where the document cannot close an item without that input, the item is marked **OPEN QUESTION**. + +--- + +## Abstract + +XH-1 is a custom 128-core RISC-V design. The verification burden scales sub-linearly with core count only if the per-core verification collateral is parameterized and the cross-core interactions (coherence, interrupts, debug, trace) are explicitly delegated to a separate multicore verification document. This document proposes a hybrid per-core methodology: constrained-random SystemVerilog/UVM simulation as the workhorse, formal property verification on targeted sub-modules (CSR file, MMU/TLB, exception priority encoders, ROB/scoreboard, FSMs), co-simulation against an architectural reference model (Spike-class), and riscv-arch-test signature generation. Coverage exit criteria are defined in terms of legal-bin functional coverage, CSR-bit and WARL coverage, exception/interrupt priority coverage, and targeted formal unbounded proofs. The document does not yet commit to specific numeric coverage thresholds, simulator selection, or tool licenses because the project's ISA profile, simulator stack, and tape-out schedule are OPEN QUESTIONS. + +--- + +## Research Question + +The specific research questions addressed are: + +1. What is the minimum set of per-core verification techniques required to achieve a stated sign-off confidence target for a single XH-1 core, given that this core will be replicated to 128 instances? +2. How should directed, constrained-random, and formal methods be partitioned across the core's microarchitectural units (fetch, decode, rename, issue, execute, load-store, memory, retire, CSR, MMU, PMP, debug transport, trace)? +3. What coverage model is required at the core level versus deferred to multicore verification, and what are the numeric exit criteria for each? +4. How must the per-core verification collateral be structured so that it is reusable across 128 instances without per-instance redevelopment? + +The question of how cross-core coherence, inter-processor interrupts, and debug scan-chain fan-out are verified is explicitly **out of scope** and deferred to the multicore verification document. + +--- + +## Background + +**FACT** — The RISC-V ISA is defined by two principal specifications maintained by RISC-V International: the *Unprivileged ISA Specification* and the *Privileged ISA Specification*. Both are ratified, versioned, and publicly available. [Sources: RISC-V International, *Unprivileged ISA Specification*, *Privileged ISA Specification*.] + +**FACT** — The RISC-V *Debug Specification* is at version 1.0 (ratified); the *Trace Specification* is at version 2.0.0 (as of the most recent public release prior to this document). [Sources: RISC-V International, *Debug Spec 1.0*, *Trace Spec 2.0.0*.] + +**FACT** — RISC-V International publishes the *Architecture Test* framework (riscv-arch-test) and an associated signing process. Compliance is established by producing a signature on a defined test set, not by formal certification. [Source: github.com/riscv-arch-test, RISC-V International *Compliance Policy*.] + +**FACT** — OpenHW Group's CV32E40P is a 32-bit, in-order, RV32IMFC RISC-V core; CV32E40PX is a separate variant that adds the XPulpV2 extension set (including the FPU) and is targeted at DSP/ML workloads. The two are not interchangeable and should not be cited as a single baseline. [Source: OpenHW Group CV32E40P and CV32E40PX core documentation.] + +**ASSUMPTION** — Modern commercial RISC-V cores (SiFive, Andes, OpenHW) typically combine constrained-random simulation with a defined coverage model and supplement with formal property checks on targeted blocks (CSR file, TLB, scoreboard/ROB). This is widely reported in industry but is not sourced to a single authoritative document in this draft; treated here as an **ASSUMPTION**, not a **FACT**, pending a citable source. + +**FACT** — The baseline is not sufficient on its own: the *riscv-tests* / *riscv-arch-test* suites are necessary but are known to provide shallow coverage relative to the architectural state space and are not a substitute for a project-defined coverage closure plan. [Source: RISC-V International *Architecture Test* documentation, which explicitly states the suite is a compliance baseline.] + +--- + +## Existing Approaches + +**FACT** — The following categories of core-level verification flow are established in the EDA and RISC-V ecosystems: + +1. Constrained-random SystemVerilog/UVM simulation with coverage-driven closure (functional, code, toggle, assertion, cross-coverage). +2. Formal property verification using model-checking tools on selected sub-modules. Correctly categorized tools in this space include: Siemens EDA JasperGold (historically branded Cadence Jasper), Cadence Jasper, Synopsys VC Formal, and OneSpin 360 DV. [Source: vendor product documentation; see correction note below.] +3. FPGA prototyping and emulation acceleration (e.g., Synopsys ZeBu, Cadence Palladium, in-house FPGA boards). +4. Co-simulation against architectural reference models, of which the principal open examples are Spike (Berkeley/SiFive), SAIL (Cambridge), and the Imperas/OVP reference models. [Sources: github.com/riscv/riscv-isa-sim (Spike); github.com/rems-project/sail-riscv; github.com/riscv-ovpsim.] +5. Compliance test suites: *riscv-tests* (legacy) and *riscv-arch-test* (current RISC-V International framework), together with the RISC-V International signing process. [Source: github.com/riscv-arch-test.] +6. Mutation-based verification tools (e.g., Certitude, proprietary in-house mutation frameworks). +7. Coverage merging across simulators and prototypes (e.g., Synopsys VCS + ZeBu, Cadence Xcelium + Palladium). + +**Correction to prior draft:** The previous version of this document listed "Synopsys Formality" under formal property verification. Formality is a sequential logic-equivalence checker (LEC) used for RTL-to-gate equivalence, not a model-checking property verification tool, and the reference was removed. The previous version also listed "JasperGold" and "Cadence Jasper" as if they were two separate products. JasperGold is the current product line (Siemens EDA; historically Cadence); the two are the same product family under different corporate brandings and are listed here as one entry with the historical name in parentheses. No fabricated evidence is introduced. + +--- + +## Alternative Designs + +The principal methodology alternatives considered are: + +- **A1. Pure simulation-driven sign-off.** Constrained-random UVM with full coverage closure; no formal. Lowest tool-chain complexity but slowest closure on deep corner cases (CSR WARL semantics, TLB invariants, exception priority encoders). +- **A2. Hybrid simulation + formal (selected).** Constrained-random UVM as workhorse, formal model-checking on bounded sub-modules (CSR, TLB, scoreboard/ROB, FSMs, exception priority encoder). Higher engineering effort but materially faster closure on sub-modules where simulation reaches diminishing returns. +- **A3. Formal-first, simulation-supplemental.** Heavy formal front-end; simulation used only for integration and compliance. Generally impractical for a complex out-of-order core (see Implementation Considerations). +- **A4. Emulation/FPGA-first.** Boots real software (Linux, distro, stress workloads) early. High infrastructure cost; visibility is limited; sign-off against the microarchitectural coverage model is still owed. + +Within the simulation domain: + +- **B1. In-house UVM testbench.** Full control, no license cost, but high engineering effort. +- **B2. Commercial verification IP.** Faster start; recurring license cost; opaque to RTL changes. +- **B3. Generator-based flows (e.g., riscv-dv from Google/CHIPS Alliance).** Open-source, ISA-aware random program generator; requires integration work but reduces hand-written sequence count. [Source: github.com/chipsalliance/riscv-dv.] + +Coverage-model alternatives: + +- **C1. Structural only (line/branch/toggle).** Fast to collect; known to miss functional holes. +- **C2. Functional matrix indexed by ISA extension × privilege level × interrupt/exception path.** Necessary for sign-off; higher collection and debug cost. +- **C3. Functional + mutation score.** Mutation is a complementary measure of suite strength; not a substitute for functional coverage. + +--- + +## Comparison + +**INSUFFICIENT EVIDENCE** for a quantitative comparison. The relative run time, license cost, engineering effort, coverage-closure time, and debug productivity of A1–A4 and B1–B3 depend on the chosen EDA stack, the ISA profile, and the in-house team's existing code base, none of which are captured in the project repository. A numeric trade-off table is not fabricated here. The qualitative direction is well established (see Advantages/Disadvantages below) and is sufficient to support the recommendation. + +--- + +## Advantages + +**Hybrid simulation + formal (A2):** + +- Formal closes deep corner cases (CSR WARL, TLB invariants, exception priority encoder) faster than simulation, where simulation reaches diminishing returns. (Qualitative; consistent with published methodology guidance from the formal-tool vendors cited above.) +- Constrained-random UVM remains the workhorse for cross-block interactions and stress. +- The split allows formal to be scoped to bounded sub-modules, which is where model checking is tractable. + +**Generator-based flow (B3, e.g., riscv-dv):** + +- Open-source and ISA-aware; reduces hand-written sequence count. +- Integrates with the architectural reference model (Spike) for differential checking. [Source: github.com/chipsalliance/riscv-dv; github.com/riscv/riscv-isa-sim.] + +**Compliance framework:** + +- riscv-arch-test with the RISC-V International signing process produces a portable, vendor-neutral artifact that downstream software/OS teams can use. [Source: github.com/riscv-arch-test; RISC-V International *Compliance Policy*.] + +--- + +## Disadvantages + +**Hybrid simulation + formal (A2):** + +- Two tool flows must be integrated, which adds engineering effort and requires a coverage-merging strategy across flows. +- Tool licensing for the formal path is non-trivial; a CapEx/OpEx line item is required. + +**Generator-based flow (B3):** + +- riscv-dv is RV32/RV64 ISA-focused; non-ISA RTL (custom accelerators, custom CSRs) is not generated and must be hand-written. + +**Compliance framework:** + +- The riscv-arch-test suite is shallow; producing a signature on it is necessary but not sufficient for sign-off. Treating it as sufficient is a known failure mode. + +**General across alternatives:** + +- None of the alternatives in isolation produces a tape-out-ready sign-off artifact; all require an explicit, project-owned coverage-closure plan with numeric exit criteria. + +--- + +## XH-1 Considerations + +**ASSUMPTION** — XH-1 is a custom 128-core RISC-V design. Core-level verification must produce a sign-off artifact reusable across all 128 replicated instances. + +**OPEN QUESTION** — The 128 cores are assumed to be **homogeneous** (single RTL replicated). Whether this holds, or whether XH-1 includes variants (e.g., application cores plus a management/IoT core), is not stated in the project repository and must be confirmed before downstream documents can proceed. + +**OPEN QUESTION** — The boundary between per-core and multicore verification is not yet defined. Candidates: + +- Per-core: L1 I-cache, L1 D-cache, MMU/TLB, PMP, CSR, debug transport module, trace encoder, retire/ROB, exception/interrupt controller. +- Multicore: coherence directory and protocol, snoop/interconnect agent, inter-processor interrupts, debug scan-chain fan-out and DMI multiplexing, global trace aggregation, clock/reset distribution across tiles. + +**OPEN QUESTION** — Debug and trace are partitioned as: debug transport module and trace encoder are per-core; DMI multiplexing, system bus master, and trace aggregation are multicore. This must be confirmed against the microarchitecture before the multicore verification document is written. + +**OPEN QUESTION** — The RISC-V Debug Specification version that applies to XH-1 is not stated; both 0.13 and 1.0 are observed in deployed cores. The Trace Specification version is similarly unresolved. Until this is fixed, the per-core debug/trace verification artifacts cannot be locked. + +--- + +## 128-Core Scalability + +**PROPOSAL** — Per-core verification collateral (testbench, sequences, assertions, coverage models) is parameterized by a *core index* / *instance ID* handle so that the same collateral validates every replicated instance. This is trivial for the simulation case and only marginally more involved for formal (where the instance ID is exposed as a constant or free variable) and for emulation (where the same image boots all 128 cores). + +**ASSUMPTION** — The 128-core design reuses a single verified core RTL. If the array is heterogeneous, the parameterization collapses to one parameterization per variant, and the per-core effort is multiplied by the number of variants. + +**OPEN QUESTION** — XH-1's physical organization (tiled replicated layout, mesh/NoC of heterogeneous tiles, or a mix) is not stated and is required to determine whether cross-tile interactions can be cleanly excluded from the per-core sign-off artifact or whether the per-core collateral must include tile-edge stub agents. + +**PROPOSAL** — Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out) are **out of scope** for the per-core document and are owned by the multicore verification document. Including them in the per-core sign-off is a known anti-pattern that produces a fragile, instance-index-coupled artifact. + +--- + +## Performance Considerations + +**OPEN QUESTION** — Core-level simulation throughput (instructions/second, coverage points/hour) depends on the chosen simulator (commercial event-driven, Verilator, cocotb, or custom), the workload, and the host farm, none of which are fixed. No numeric simulation throughput is fabricated here. + +**ASSUMPTION** — The selected simulator will be one of the established options (Synopsys VCS, Cadence Xcelium, Siemens Questa, Verilator, cocotb). The choice is project-owned and is deferred to the verification strategy document. + +**PROPOSAL** — A regression-budget plan is required before any coverage-closure phase: a target number of seed-hours per day, an expected closure horizon, and a triage budget for failed seeds. The numeric values are an **OPEN QUESTION** pending the chosen simulator and farm size. + +--- + +## Area Considerations + +**INSUFFICIENT EVIDENCE** — Verification-influenced area decisions (adding observability or assertion logic, formal-friendly state encodings, scan/DFT hooks) are not characterized for XH-1. + +**ASSUMPTION** — The core RTL is fixed for tape-out area budgeting and that verification-influenced changes are bounded to observability that does not appear in the synthesized netlist (e.g., `ifdef` guards). This must be confirmed against the synthesis/scan insertion flow. + +**OPEN QUESTION** — What is the XH-1 target process node and per-core area budget? Both are required before area-influenced verification choices (e.g., one-hot vs. binary FSM encoding for formal tractability) can be evaluated. + +--- + +## Power and Energy Considerations + +**INSUFFICIENT EVIDENCE** — Per-core power and energy are not characterized for XH-1. + +**ASSUMPTION** — Verification-influenced power decisions (e.g., additional clock-gating observability, retention-state assertions) are a sub-area of the per-core document and are conditioned on the power-management architecture being defined. + +**OPEN QUESTION** — Does XH-1 implement RISC-V power-management extensions (e.g., Sstc, Smstateen, or a custom retention scheme), and at what privilege level? Until this is answered, power-aware verification cannot be scoped. + +**OPEN QUESTION** — How is energy-aware DV (e.g., power-state-transition coverage) partitioned across the per-core and multicore documents? An explicit ownership line is required. + +--- + +## Implementation Considerations + +**ASSUMPTION** — XH-1 implements a defined subset of the RISC-V ISA (base integer, M, A, C, and a selected set of F/D, V, B, H, K, and platform extensions). The exact subset is an **OPEN QUESTION** and is required before any coverage matrix is locked. + +**OPEN QUESTION** — Is the core in-order or out-of-order? This drives the verification surface: + +- In-order: simpler scoreboard, no speculation-recovery paths. +- Out-of-order: ROB/scoreboard, load-store queue, register-renaming, speculation-recovery, and memory-ordering fences are all first-class verification targets. A formal model of the ROB is typically bounded and tractable; a full out-of-order memory subsystem is generally beyond unbounded proof. + +**ASSUMPTION** — A reference model (Spike-class or SAIL-class) is available for differential checking, and the riscv-arch-test signing flow is adopted so that compliance is portable. Both are working assumptions until confirmed. + +**PROPOSAL** — Where the project's extensions include V (vector), B (bitmanip), H (hypervisor), or K (cryptography/scalar crypto), the per-core coverage matrix is extended with extension-specific cross-coverage, and formal proofs on the corresponding functional units are added to the formal scope. The conditional language here replaces the previous draft's unconditional listing. + +**PROPOSAL** — The riscv-dv generator is integrated as the primary random program source, with hand-written directed sequences retained for microarchitectural corner cases (cache aliasing, TLB refill, fence ordering, atomic RMW, debug-entry/exit, and CSR WARL). [Source: github.com/chipsalliance/riscv-dv.] + +--- + +## Verification Considerations + +**PROPOSAL** — A coverage closure plan is defined before any RTL is signed off. Exit criteria are stated in terms of: + +- **Functional coverage.** Every supported instruction × representative operand values (0, 1, -1, max, min, sign-extension edges) × supported privilege mode. +- **Exception/interrupt coverage.** Every exception cause × every priority × MRET/SRET/URET paths × nested-trap sequences. [Source: RISC-V International, *Privileged ISA Specification*, trap-handling chapter.] +- **CSR coverage.** Every implemented CSR × every writable bit × reset value × WARL semantics. Formal proof of WARL invariants is a sub-goal. +- **Memory access coverage.** Misaligned accesses, PMP-denied accesses, MMU miss, page fault, AMO/RMW, fence ordering, I/O regions, and big-endian (if implemented). +- **Reset and debug.** Every reset reason (power-on, warm, retention, debug) and every debug state transition in the RISC-V Debug Specification. +- **Formal.** Targeted unbounded proofs on bounded sub-modules: CSR file WARL invariants, TLB invariants, exception priority encoder, retire/ROB, debug-transport FSM, trace-encoder FSM. + +**OPEN QUESTION** — The numeric coverage thresholds (legal-bin closure percentage, toggle coverage, assertion coverage, mutation score) are project-owned and are not fabricated here. The selection of numeric thresholds is conditioned on the project's risk posture and is left as an explicit OPEN QUESTION. + +**PROPOSAL** — Extension-specific items in the coverage list are conditioned on the confirmed ISA profile: + +- F/D coverage (including NaN-boxing) is included only if the F/D extension is implemented. +- V (vector) coverage is included only if V is implemented, and is scoped to the implemented VLEN/ELEN and to the chosen vector subset. +- B (bitmanip) coverage is included only if B is implemented. +- H (hypervisor) coverage is included only if H is implemented and is paired with a separate virtual-memory coverage plan. +- K (scalar crypto) coverage is included only if K is implemented, with side-channel considerations deferred to a separate document. + +This conditional scoping replaces the previous draft's unconditional listing of F/D-specific items. + +**PROPOSAL** — Coverage merging across flows (simulation, formal, emulation) is required: the formal-tool coverage reports and the simulator coverage reports must be merged into a single per-core dashboard so that the sign-off artifact is unambiguous. + +**OPEN QUESTION** — Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level? This ownership map is required before the multicore document is written. + +--- + +## Software Considerations + +**PROPOSAL** — The per-core verification artifacts are aligned with the downstream software flow: + +- The riscv-arch-test signature is producible from the verified core and is delivered alongside the reference manual. +- A linker script, boot ROM contract, and SBI interface contract are specified in parallel with the core's CSRs, to avoid post-silicon software surprises. [Source: RISC-V International, *SBI Specification*.] +- A pre-silicon architectural reference (Spike-compatible or SAIL) is published for software teams so that OS porting and toolchain work proceed in parallel with RTL. + +**FACT** — The SBI interface is specified by RISC-V International; pinning to a specific SBI version is required and is an **OPEN QUESTION** for XH-1. + +--- + +## Recommendation + +**RECOMMENDATION** — Adopt the hybrid simulation + formal methodology (A2) for per-core verification, with the following binding choices: + +1. **Simulation workhorse.** Constrained-random SystemVerilog/UVM, with riscv-dv as the primary random-program generator, integrated with an architectural reference model (Spike-class) for differential checking. +2. **Formal scope.** Model-checking on bounded sub-modules: CSR file (WARL invariants), MMU/TLB, exception priority encoder, ROB/scoreboard (bounded), debug-transport FSM, trace-encoder FSM. Tools are selected from the model-checking category (JasperGold, VC Formal, OneSpin 360 DV); Formality is explicitly out of scope for this document. +3. **Compliance.** riscv-arch-test signature generation is part of the sign-off artifact. +4. **Coverage model.** Functional matrix indexed by ISA extension × privilege level × interrupt/exception path, with extension-specific items conditioned on the confirmed ISA profile. +5. **Parameterization.** All per-core verification collateral is parameterized by core index / instance ID, with cross-core interactions explicitly out of scope. +6. **Ownership.** Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are owned by the multicore verification document. + +The numeric coverage thresholds, the choice of simulator and formal tool, and the coverage-closure horizon are **OPEN QUESTIONS** and are deferred to the verification strategy document and the project's sign-off policy. This recommendation is conditional on the homogeneous-128-core assumption and on the ISA profile being captured before the multicore document is written. + +--- + +## Confidence + +Medium. The methodology recommendation follows from established industry practice and the cited RISC-V specifications; the recommendation does not depend on the unresolved numeric thresholds. Confidence would be upgraded to High once the **OPEN QUESTIONS** below are closed. + +--- + +## Open Questions + +- What RISC-V base and extensions does XH-1 implement (RV32/RV64, I/M/A/F/D/C/V/B/H/K, supported privilege modes, hypervisor)? +- Is the core in-order or out-of-order? +- Is the 128-core array homogeneous, or are there variants (application cores plus a management/IoT core)? +- What is the XH-1 target process node, per-core area budget, clock frequency, and tape-out schedule? +- What simulator and formal-tool stack is available or mandated (commercial, Verilator, cocotb, custom)? +- What is the project's stance on the RISC-V International compliance signing process, and at what SBI version is the platform pinned? +- What numeric coverage exit criteria (legal-bin percentage, toggle, assertion, mutation score) does the project accept for tape-out? +- Which RISC-V Debug Specification version (0.13 or 1.0) and which Trace Specification version apply to XH-1? +- Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level? +- Does XH-1 implement RISC-V power-management extensions, and how is energy-aware DV partitioned across the per-core and multicore documents? +- Is FPGA prototyping capacity available at the scale required to boot 128 cores? + +--- + +## Sources + +- RISC-V International, *Unprivileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/ +- RISC-V International, *Privileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/ +- RISC-V International, *Debug Specification*, version 1.0. https://riscv.org/technical/specifications/ +- RISC-V International, *Trace Specification*, version 2.0.0. https://riscv.org/technical/specifications/ +- RISC-V International, *SBI Specification*. https://riscv.org/technical/specifications/ +- RISC-V International, *Compliance Policy* and *Architecture Test* framework. https://github.com/riscv-arch-test +- OpenHW Group, *CV32E40P Core Documentation* and *CV32E40PX Core Documentation*. https://github.com/openhwgroup +- Google / CHIPS Alliance, *riscv-dv* (ISA-aware random program generator). https://github.com/chipsalliance/riscv-dv +- Berkeley / SiFive, *Spike* (riscv-isa-sim). https://github.com/riscv/riscv-isa-sim +- Cambridge, *SAIL RISC-V* (architectural reference model). https://github.com/rems-project/sail-riscv +- Imperas, *riscvOVPsim* (commercial reference model). https://github.com/riscv-ovpsim +- Siemens EDA, *JasperGold* formal verification platform (product documentation; cited as the model-checking tool in scope). +- Synopsys, *VC Formal* (product documentation; cited as a model-checking tool in scope). +- OneSpin, *360 DV* (product documentation; cited as a model-checking tool in scope). + +`INSUFFICIENT EVIDENCE` is recorded above in the Performance, Area, Power and Energy, and Comparison sections pending the project inputs that close the corresponding **OPEN QUESTIONS**. No numeric claim in any of those sections has been fabricated. diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round2.md.json b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round2.md.json new file mode 100644 index 0000000..090146f --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round2.md.json @@ -0,0 +1,383 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787740176-63Wr7EQB9XC71LLuiFXv","object":"chat.completion","created":1787740176,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"# Core Verification\n\n## Document Conventions\n\nThis document uses a fixed label taxonomy applied to every claim:\n\n- **FACT** — A statement supported by a cited primary or authoritative source.\n- **ASSUMPTION** — A working hypothesis taken by the author in the absence of confirmed project data; must be re-validated before sign-off.\n- **PROPOSAL** — A concrete design or methodology recommendation that the document puts forward, conditioned on the stated assumptions.\n- **RECOMMENDATION** — A binding choice that follows from the analysis in this document, after considering the stated alternatives.\n- **OPEN QUESTION** — A question whose answer is required before the surrounding analysis can be finalized; explicitly not closed by this document.\n\nWhere evidence is not available, the document writes `INSUFFICIENT EVIDENCE` rather than fabricating data.\n\n---\n\n## Status\n\nSubstantive research draft. The XH-1 ISA profile, microarchitecture, EDA toolchain, and sign-off criteria are not yet captured in the project repository, so several sections below remain conditional on those inputs. Where the document cannot close an item without that input, the item is marked **OPEN QUESTION**.\n\n---\n\n## Abstract\n\nXH-1 is a custom 128-core RISC-V design. The verification burden scales sub-linearly with core count only if the per-core verification collateral is parameterized and the cross-core interactions (coherence, interrupts, debug, trace) are explicitly delegated to a separate multicore verification document. This document proposes a hybrid per-core methodology: constrained-random SystemVerilog/UVM simulation as the workhorse, formal property verification on targeted sub-modules (CSR file, MMU/TLB, exception priority encoders, ROB/scoreboard, FSMs), co-simulation against an architectural reference model (Spike-class), and riscv-arch-test signature generation. Coverage exit criteria are defined in terms of legal-bin functional coverage, CSR-bit and WARL coverage, exception/interrupt priority coverage, and targeted formal unbounded proofs. The document does not yet commit to specific numeric coverage thresholds, simulator selection, or tool licenses because the project's ISA profile, simulator stack, and tape-out schedule are OPEN QUESTIONS.\n\n---\n\n## Research Question\n\nThe specific research questions addressed are:\n\n1. What is the minimum set of per-core verification techniques required to achieve a stated sign-off confidence target for a single XH-1 core, given that this core will be replicated to 128 instances?\n2. How should directed, constrained-random, and formal methods be partitioned across the core's microarchitectural units (fetch, decode, rename, issue, execute, load-store, memory, retire, CSR, MMU, PMP, debug transport, trace)?\n3. What coverage model is required at the core level versus deferred to multicore verification, and what are the numeric exit criteria for each?\n4. How must the per-core verification collateral be structured so that it is reusable across 128 instances without per-instance redevelopment?\n\nThe question of how cross-core coherence, inter-processor interrupts, and debug scan-chain fan-out are verified is explicitly **out of scope** and deferred to the multicore verification document.\n\n---\n\n## Background\n\n**FACT** — The RISC-V ISA is defined by two principal specifications maintained by RISC-V International: the *Unprivileged ISA Specification* and the *Privileged ISA Specification*. Both are ratified, versioned, and publicly available. [Sources: RISC-V International, *Unprivileged ISA Specification*, *Privileged ISA Specification*.]\n\n**FACT** — The RISC-V *Debug Specification* is at version 1.0 (ratified); the *Trace Specification* is at version 2.0.0 (as of the most recent public release prior to this document). [Sources: RISC-V International, *Debug Spec 1.0*, *Trace Spec 2.0.0*.]\n\n**FACT** — RISC-V International publishes the *Architecture Test* framework (riscv-arch-test) and an associated signing process. Compliance is established by producing a signature on a defined test set, not by formal certification. [Source: github.com/riscv-arch-test, RISC-V International *Compliance Policy*.]\n\n**FACT** — OpenHW Group's CV32E40P is a 32-bit, in-order, RV32IMFC RISC-V core; CV32E40PX is a separate variant that adds the XPulpV2 extension set (including the FPU) and is targeted at DSP/ML workloads. The two are not interchangeable and should not be cited as a single baseline. [Source: OpenHW Group CV32E40P and CV32E40PX core documentation.]\n\n**ASSUMPTION** — Modern commercial RISC-V cores (SiFive, Andes, OpenHW) typically combine constrained-random simulation with a defined coverage model and supplement with formal property checks on targeted blocks (CSR file, TLB, scoreboard/ROB). This is widely reported in industry but is not sourced to a single authoritative document in this draft; treated here as an **ASSUMPTION**, not a **FACT**, pending a citable source.\n\n**FACT** — The baseline is not sufficient on its own: the *riscv-tests* / *riscv-arch-test* suites are necessary but are known to provide shallow coverage relative to the architectural state space and are not a substitute for a project-defined coverage closure plan. [Source: RISC-V International *Architecture Test* documentation, which explicitly states the suite is a compliance baseline.]\n\n---\n\n## Existing Approaches\n\n**FACT** — The following categories of core-level verification flow are established in the EDA and RISC-V ecosystems:\n\n1. Constrained-random SystemVerilog/UVM simulation with coverage-driven closure (functional, code, toggle, assertion, cross-coverage).\n2. Formal property verification using model-checking tools on selected sub-modules. Correctly categorized tools in this space include: Siemens EDA JasperGold (historically branded Cadence Jasper), Cadence Jasper, Synopsys VC Formal, and OneSpin 360 DV. [Source: vendor product documentation; see correction note below.]\n3. FPGA prototyping and emulation acceleration (e.g., Synopsys ZeBu, Cadence Palladium, in-house FPGA boards).\n4. Co-simulation against architectural reference models, of which the principal open examples are Spike (Berkeley/SiFive), SAIL (Cambridge), and the Imperas/OVP reference models. [Sources: github.com/riscv/riscv-isa-sim (Spike); github.com/rems-project/sail-riscv; github.com/riscv-ovpsim.]\n5. Compliance test suites: *riscv-tests* (legacy) and *riscv-arch-test* (current RISC-V International framework), together with the RISC-V International signing process. [Source: github.com/riscv-arch-test.]\n6. Mutation-based verification tools (e.g., Certitude, proprietary in-house mutation frameworks).\n7. Coverage merging across simulators and prototypes (e.g., Synopsys VCS + ZeBu, Cadence Xcelium + Palladium).\n\n**Correction to prior draft:** The previous version of this document listed \"Synopsys Formality\" under formal property verification. Formality is a sequential logic-equivalence checker (LEC) used for RTL-to-gate equivalence, not a model-checking property verification tool, and the reference was removed. The previous version also listed \"JasperGold\" and \"Cadence Jasper\" as if they were two separate products. JasperGold is the current product line (Siemens EDA; historically Cadence); the two are the same product family under different corporate brandings and are listed here as one entry with the historical name in parentheses. No fabricated evidence is introduced.\n\n---\n\n## Alternative Designs\n\nThe principal methodology alternatives considered are:\n\n- **A1. Pure simulation-driven sign-off.** Constrained-random UVM with full coverage closure; no formal. Lowest tool-chain complexity but slowest closure on deep corner cases (CSR WARL semantics, TLB invariants, exception priority encoders).\n- **A2. Hybrid simulation + formal (selected).** Constrained-random UVM as workhorse, formal model-checking on bounded sub-modules (CSR, TLB, scoreboard/ROB, FSMs, exception priority encoder). Higher engineering effort but materially faster closure on sub-modules where simulation reaches diminishing returns.\n- **A3. Formal-first, simulation-supplemental.** Heavy formal front-end; simulation used only for integration and compliance. Generally impractical for a complex out-of-order core (see Implementation Considerations).\n- **A4. Emulation/FPGA-first.** Boots real software (Linux, distro, stress workloads) early. High infrastructure cost; visibility is limited; sign-off against the microarchitectural coverage model is still owed.\n\nWithin the simulation domain:\n\n- **B1. In-house UVM testbench.** Full control, no license cost, but high engineering effort.\n- **B2. Commercial verification IP.** Faster start; recurring license cost; opaque to RTL changes.\n- **B3. Generator-based flows (e.g., riscv-dv from Google/CHIPS Alliance).** Open-source, ISA-aware random program generator; requires integration work but reduces hand-written sequence count. [Source: github.com/chipsalliance/riscv-dv.]\n\nCoverage-model alternatives:\n\n- **C1. Structural only (line/branch/toggle).** Fast to collect; known to miss functional holes.\n- **C2. Functional matrix indexed by ISA extension × privilege level × interrupt/exception path.** Necessary for sign-off; higher collection and debug cost.\n- **C3. Functional + mutation score.** Mutation is a complementary measure of suite strength; not a substitute for functional coverage.\n\n---\n\n## Comparison\n\n**INSUFFICIENT EVIDENCE** for a quantitative comparison. The relative run time, license cost, engineering effort, coverage-closure time, and debug productivity of A1–A4 and B1–B3 depend on the chosen EDA stack, the ISA profile, and the in-house team's existing code base, none of which are captured in the project repository. A numeric trade-off table is not fabricated here. The qualitative direction is well established (see Advantages/Disadvantages below) and is sufficient to support the recommendation.\n\n---\n\n## Advantages\n\n**Hybrid simulation + formal (A2):**\n\n- Formal closes deep corner cases (CSR WARL, TLB invariants, exception priority encoder) faster than simulation, where simulation reaches diminishing returns. (Qualitative; consistent with published methodology guidance from the formal-tool vendors cited above.)\n- Constrained-random UVM remains the workhorse for cross-block interactions and stress.\n- The split allows formal to be scoped to bounded sub-modules, which is where model checking is tractable.\n\n**Generator-based flow (B3, e.g., riscv-dv):**\n\n- Open-source and ISA-aware; reduces hand-written sequence count.\n- Integrates with the architectural reference model (Spike) for differential checking. [Source: github.com/chipsalliance/riscv-dv; github.com/riscv/riscv-isa-sim.]\n\n**Compliance framework:**\n\n- riscv-arch-test with the RISC-V International signing process produces a portable, vendor-neutral artifact that downstream software/OS teams can use. [Source: github.com/riscv-arch-test; RISC-V International *Compliance Policy*.]\n\n---\n\n## Disadvantages\n\n**Hybrid simulation + formal (A2):**\n\n- Two tool flows must be integrated, which adds engineering effort and requires a coverage-merging strategy across flows.\n- Tool licensing for the formal path is non-trivial; a CapEx/OpEx line item is required.\n\n**Generator-based flow (B3):**\n\n- riscv-dv is RV32/RV64 ISA-focused; non-ISA RTL (custom accelerators, custom CSRs) is not generated and must be hand-written.\n\n**Compliance framework:**\n\n- The riscv-arch-test suite is shallow; producing a signature on it is necessary but not sufficient for sign-off. Treating it as sufficient is a known failure mode.\n\n**General across alternatives:**\n\n- None of the alternatives in isolation produces a tape-out-ready sign-off artifact; all require an explicit, project-owned coverage-closure plan with numeric exit criteria.\n\n---\n\n## XH-1 Considerations\n\n**ASSUMPTION** — XH-1 is a custom 128-core RISC-V design. Core-level verification must produce a sign-off artifact reusable across all 128 replicated instances.\n\n**OPEN QUESTION** — The 128 cores are assumed to be **homogeneous** (single RTL replicated). Whether this holds, or whether XH-1 includes variants (e.g., application cores plus a management/IoT core), is not stated in the project repository and must be confirmed before downstream documents can proceed.\n\n**OPEN QUESTION** — The boundary between per-core and multicore verification is not yet defined. Candidates:\n\n- Per-core: L1 I-cache, L1 D-cache, MMU/TLB, PMP, CSR, debug transport module, trace encoder, retire/ROB, exception/interrupt controller.\n- Multicore: coherence directory and protocol, snoop/interconnect agent, inter-processor interrupts, debug scan-chain fan-out and DMI multiplexing, global trace aggregation, clock/reset distribution across tiles.\n\n**OPEN QUESTION** — Debug and trace are partitioned as: debug transport module and trace encoder are per-core; DMI multiplexing, system bus master, and trace aggregation are multicore. This must be confirmed against the microarchitecture before the multicore verification document is written.\n\n**OPEN QUESTION** — The RISC-V Debug Specification version that applies to XH-1 is not stated; both 0.13 and 1.0 are observed in deployed cores. The Trace Specification version is similarly unresolved. Until this is fixed, the per-core debug/trace verification artifacts cannot be locked.\n\n---\n\n## 128-Core Scalability\n\n**PROPOSAL** — Per-core verification collateral (testbench, sequences, assertions, coverage models) is parameterized by a *core index* / *instance ID* handle so that the same collateral validates every replicated instance. This is trivial for the simulation case and only marginally more involved for formal (where the instance ID is exposed as a constant or free variable) and for emulation (where the same image boots all 128 cores).\n\n**ASSUMPTION** — The 128-core design reuses a single verified core RTL. If the array is heterogeneous, the parameterization collapses to one parameterization per variant, and the per-core effort is multiplied by the number of variants.\n\n**OPEN QUESTION** — XH-1's physical organization (tiled replicated layout, mesh/NoC of heterogeneous tiles, or a mix) is not stated and is required to determine whether cross-tile interactions can be cleanly excluded from the per-core sign-off artifact or whether the per-core collateral must include tile-edge stub agents.\n\n**PROPOSAL** — Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out) are **out of scope** for the per-core document and are owned by the multicore verification document. Including them in the per-core sign-off is a known anti-pattern that produces a fragile, instance-index-coupled artifact.\n\n---\n\n## Performance Considerations\n\n**OPEN QUESTION** — Core-level simulation throughput (instructions/second, coverage points/hour) depends on the chosen simulator (commercial event-driven, Verilator, cocotb, or custom), the workload, and the host farm, none of which are fixed. No numeric simulation throughput is fabricated here.\n\n**ASSUMPTION** — The selected simulator will be one of the established options (Synopsys VCS, Cadence Xcelium, Siemens Questa, Verilator, cocotb). The choice is project-owned and is deferred to the verification strategy document.\n\n**PROPOSAL** — A regression-budget plan is required before any coverage-closure phase: a target number of seed-hours per day, an expected closure horizon, and a triage budget for failed seeds. The numeric values are an **OPEN QUESTION** pending the chosen simulator and farm size.\n\n---\n\n## Area Considerations\n\n**INSUFFICIENT EVIDENCE** — Verification-influenced area decisions (adding observability or assertion logic, formal-friendly state encodings, scan/DFT hooks) are not characterized for XH-1.\n\n**ASSUMPTION** — The core RTL is fixed for tape-out area budgeting and that verification-influenced changes are bounded to observability that does not appear in the synthesized netlist (e.g., `ifdef` guards). This must be confirmed against the synthesis/scan insertion flow.\n\n**OPEN QUESTION** — What is the XH-1 target process node and per-core area budget? Both are required before area-influenced verification choices (e.g., one-hot vs. binary FSM encoding for formal tractability) can be evaluated.\n\n---\n\n## Power and Energy Considerations\n\n**INSUFFICIENT EVIDENCE** — Per-core power and energy are not characterized for XH-1.\n\n**ASSUMPTION** — Verification-influenced power decisions (e.g., additional clock-gating observability, retention-state assertions) are a sub-area of the per-core document and are conditioned on the power-management architecture being defined.\n\n**OPEN QUESTION** — Does XH-1 implement RISC-V power-management extensions (e.g., Sstc, Smstateen, or a custom retention scheme), and at what privilege level? Until this is answered, power-aware verification cannot be scoped.\n\n**OPEN QUESTION** — How is energy-aware DV (e.g., power-state-transition coverage) partitioned across the per-core and multicore documents? An explicit ownership line is required.\n\n---\n\n## Implementation Considerations\n\n**ASSUMPTION** — XH-1 implements a defined subset of the RISC-V ISA (base integer, M, A, C, and a selected set of F/D, V, B, H, K, and platform extensions). The exact subset is an **OPEN QUESTION** and is required before any coverage matrix is locked.\n\n**OPEN QUESTION** — Is the core in-order or out-of-order? This drives the verification surface:\n\n- In-order: simpler scoreboard, no speculation-recovery paths.\n- Out-of-order: ROB/scoreboard, load-store queue, register-renaming, speculation-recovery, and memory-ordering fences are all first-class verification targets. A formal model of the ROB is typically bounded and tractable; a full out-of-order memory subsystem is generally beyond unbounded proof.\n\n**ASSUMPTION** — A reference model (Spike-class or SAIL-class) is available for differential checking, and the riscv-arch-test signing flow is adopted so that compliance is portable. Both are working assumptions until confirmed.\n\n**PROPOSAL** — Where the project's extensions include V (vector), B (bitmanip), H (hypervisor), or K (cryptography/scalar crypto), the per-core coverage matrix is extended with extension-specific cross-coverage, and formal proofs on the corresponding functional units are added to the formal scope. The conditional language here replaces the previous draft's unconditional listing.\n\n**PROPOSAL** — The riscv-dv generator is integrated as the primary random program source, with hand-written directed sequences retained for microarchitectural corner cases (cache aliasing, TLB refill, fence ordering, atomic RMW, debug-entry/exit, and CSR WARL). [Source: github.com/chipsalliance/riscv-dv.]\n\n---\n\n## Verification Considerations\n\n**PROPOSAL** — A coverage closure plan is defined before any RTL is signed off. Exit criteria are stated in terms of:\n\n- **Functional coverage.** Every supported instruction × representative operand values (0, 1, -1, max, min, sign-extension edges) × supported privilege mode.\n- **Exception/interrupt coverage.** Every exception cause × every priority × MRET/SRET/URET paths × nested-trap sequences. [Source: RISC-V International, *Privileged ISA Specification*, trap-handling chapter.]\n- **CSR coverage.** Every implemented CSR × every writable bit × reset value × WARL semantics. Formal proof of WARL invariants is a sub-goal.\n- **Memory access coverage.** Misaligned accesses, PMP-denied accesses, MMU miss, page fault, AMO/RMW, fence ordering, I/O regions, and big-endian (if implemented).\n- **Reset and debug.** Every reset reason (power-on, warm, retention, debug) and every debug state transition in the RISC-V Debug Specification.\n- **Formal.** Targeted unbounded proofs on bounded sub-modules: CSR file WARL invariants, TLB invariants, exception priority encoder, retire/ROB, debug-transport FSM, trace-encoder FSM.\n\n**OPEN QUESTION** — The numeric coverage thresholds (legal-bin closure percentage, toggle coverage, assertion coverage, mutation score) are project-owned and are not fabricated here. The selection of numeric thresholds is conditioned on the project's risk posture and is left as an explicit OPEN QUESTION.\n\n**PROPOSAL** — Extension-specific items in the coverage list are conditioned on the confirmed ISA profile:\n\n- F/D coverage (including NaN-boxing) is included only if the F/D extension is implemented.\n- V (vector) coverage is included only if V is implemented, and is scoped to the implemented VLEN/ELEN and to the chosen vector subset.\n- B (bitmanip) coverage is included only if B is implemented.\n- H (hypervisor) coverage is included only if H is implemented and is paired with a separate virtual-memory coverage plan.\n- K (scalar crypto) coverage is included only if K is implemented, with side-channel considerations deferred to a separate document.\n\nThis conditional scoping replaces the previous draft's unconditional listing of F/D-specific items.\n\n**PROPOSAL** — Coverage merging across flows (simulation, formal, emulation) is required: the formal-tool coverage reports and the simulator coverage reports must be merged into a single per-core dashboard so that the sign-off artifact is unambiguous.\n\n**OPEN QUESTION** — Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level? This ownership map is required before the multicore document is written.\n\n---\n\n## Software Considerations\n\n**PROPOSAL** — The per-core verification artifacts are aligned with the downstream software flow:\n\n- The riscv-arch-test signature is producible from the verified core and is delivered alongside the reference manual.\n- A linker script, boot ROM contract, and SBI interface contract are specified in parallel with the core's CSRs, to avoid post-silicon software surprises. [Source: RISC-V International, *SBI Specification*.]\n- A pre-silicon architectural reference (Spike-compatible or SAIL) is published for software teams so that OS porting and toolchain work proceed in parallel with RTL.\n\n**FACT** — The SBI interface is specified by RISC-V International; pinning to a specific SBI version is required and is an **OPEN QUESTION** for XH-1.\n\n---\n\n## Recommendation\n\n**RECOMMENDATION** — Adopt the hybrid simulation + formal methodology (A2) for per-core verification, with the following binding choices:\n\n1. **Simulation workhorse.** Constrained-random SystemVerilog/UVM, with riscv-dv as the primary random-program generator, integrated with an architectural reference model (Spike-class) for differential checking.\n2. **Formal scope.** Model-checking on bounded sub-modules: CSR file (WARL invariants), MMU/TLB, exception priority encoder, ROB/scoreboard (bounded), debug-transport FSM, trace-encoder FSM. Tools are selected from the model-checking category (JasperGold, VC Formal, OneSpin 360 DV); Formality is explicitly out of scope for this document.\n3. **Compliance.** riscv-arch-test signature generation is part of the sign-off artifact.\n4. **Coverage model.** Functional matrix indexed by ISA extension × privilege level × interrupt/exception path, with extension-specific items conditioned on the confirmed ISA profile.\n5. **Parameterization.** All per-core verification collateral is parameterized by core index / instance ID, with cross-core interactions explicitly out of scope.\n6. **Ownership.** Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are owned by the multicore verification document.\n\nThe numeric coverage thresholds, the choice of simulator and formal tool, and the coverage-closure horizon are **OPEN QUESTIONS** and are deferred to the verification strategy document and the project's sign-off policy. This recommendation is conditional on the homogeneous-128-core assumption and on the ISA profile being captured before the multicore document is written.\n\n---\n\n## Confidence\n\nMedium. The methodology recommendation follows from established industry practice and the cited RISC-V specifications; the recommendation does not depend on the unresolved numeric thresholds. Confidence would be upgraded to High once the **OPEN QUESTIONS** below are closed.\n\n---\n\n## Open Questions\n\n- What RISC-V base and extensions does XH-1 implement (RV32/RV64, I/M/A/F/D/C/V/B/H/K, supported privilege modes, hypervisor)?\n- Is the core in-order or out-of-order?\n- Is the 128-core array homogeneous, or are there variants (application cores plus a management/IoT core)?\n- What is the XH-1 target process node, per-core area budget, clock frequency, and tape-out schedule?\n- What simulator and formal-tool stack is available or mandated (commercial, Verilator, cocotb, custom)?\n- What is the project's stance on the RISC-V International compliance signing process, and at what SBI version is the platform pinned?\n- What numeric coverage exit criteria (legal-bin percentage, toggle, assertion, mutation score) does the project accept for tape-out?\n- Which RISC-V Debug Specification version (0.13 or 1.0) and which Trace Specification version apply to XH-1?\n- Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level?\n- Does XH-1 implement RISC-V power-management extensions, and how is energy-aware DV partitioned across the per-core and multicore documents?\n- Is FPGA prototyping capacity available at the scale required to boot 128 cores?\n\n---\n\n## Sources\n\n- RISC-V International, *Unprivileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/\n- RISC-V International, *Privileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/\n- RISC-V International, *Debug Specification*, version 1.0. https://riscv.org/technical/specifications/\n- RISC-V International, *Trace Specification*, version 2.0.0. https://riscv.org/technical/specifications/\n- RISC-V International, *SBI Specification*. https://riscv.org/technical/specifications/\n- RISC-V International, *Compliance Policy* and *Architecture Test* framework. https://github.com/riscv-arch-test\n- OpenHW Group, *CV32E40P Core Documentation* and *CV32E40PX Core Documentation*. https://github.com/openhwgroup\n- Google / CHIPS Alliance, *riscv-dv* (ISA-aware random program generator). https://github.com/chipsalliance/riscv-dv\n- Berkeley / SiFive, *Spike* (riscv-isa-sim). https://github.com/riscv/riscv-isa-sim\n- Cambridge, *SAIL RISC-V* (architectural reference model). https://github.com/rems-project/sail-riscv\n- Imperas, *riscvOVPsim* (commercial reference model). https://github.com/riscv-ovpsim\n- Siemens EDA, *JasperGold* formal verification platform (product documentation; cited as the model-checking tool in scope).\n- Synopsys, *VC Formal* (product documentation; cited as a model-checking tool in scope).\n- OneSpin, *360 DV* (product documentation; cited as a model-checking tool in scope).\n\n`INSUFFICIENT EVIDENCE` is recorded above in the Performance, Area, Power and Energy, and Comparison sections pending the project inputs that close the corresponding **OPEN QUESTIONS**. No numeric claim in any of those sections has been fabricated.","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":3370,"completion_tokens":5815,"total_tokens":9185,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":135,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round3.md b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round3.md new file mode 100644 index 0000000..4e3fe91 --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round3.md @@ -0,0 +1,353 @@ +# Core Verification + +## Document Conventions + +This document uses a fixed label taxonomy applied to every claim: + +- **FACT** — A statement supported by a cited primary or authoritative source. +- **ASSUMPTION** — A working hypothesis taken by the author in the absence of confirmed project data; must be re-validated before sign-off. +- **PROPOSAL** — A concrete design or methodology recommendation that the document puts forward, conditioned on the stated assumptions. +- **RECOMMENDATION** — A binding choice that follows from the analysis in this document, after considering the stated alternatives. +- **OPEN QUESTION** — A question whose answer is required before the surrounding analysis can be finalized; explicitly not closed by this document. + +Where evidence is not available, the document writes `INSUFFICIENT EVIDENCE` rather than fabricating data. + +--- + +## Status + +Substantive research draft. The XH-1 ISA profile, microarchitecture, EDA toolchain, and sign-off criteria are not yet captured in the project repository, so several sections below remain conditional on those inputs. Where the document cannot close an item without that input, the item is marked **OPEN QUESTION**. + +--- + +## Abstract + +XH-1 is a custom 128-core RISC-V design. The verification burden scales sub-linearly with core count only if the per-core verification collateral is instance-independent (a single testbench and coverage model reused for every replicated instance) and the cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are explicitly delegated to a separate multicore verification document. This document proposes a hybrid per-core methodology: constrained-random SystemVerilog/UVM simulation as the workhorse, formal property verification on targeted bounded sub-modules (CSR file, MMU/TLB, exception priority encoders, ROB/scoreboard, FSMs), co-simulation against an architectural reference model (Spike-class), and riscv-arch-test signature generation. Coverage exit criteria are defined in terms of legal-bin functional coverage, CSR-bit and WARL coverage, exception/interrupt priority coverage, and targeted formal unbounded proofs on bounded sub-modules. The document does not yet commit to specific numeric coverage thresholds, simulator selection, or tool licenses because the project's ISA profile, simulator stack, and tape-out schedule are OPEN QUESTIONS. + +--- + +## Research Question + +The specific research questions addressed are: + +1. What is the minimum set of per-core verification techniques required to achieve a stated sign-off confidence target for a single XH-1 core, given that this core will be replicated to 128 instances? +2. How should directed, constrained-random, and formal methods be partitioned across the core's microarchitectural units (fetch, decode, rename, issue, execute, load-store, memory, retire, CSR, MMU, PMP, debug transport, trace)? +3. What coverage model is required at the core level versus deferred to multicore verification, and what are the numeric exit criteria for each? +4. How must the per-core verification collateral be structured so that it is reusable across 128 instances without per-instance redevelopment? + +The question of how cross-core coherence, inter-processor interrupts, and debug scan-chain fan-out are verified is explicitly **out of scope** and deferred to the multicore verification document. + +--- + +## Background + +**FACT** — The RISC-V ISA is defined by two principal specifications maintained by RISC-V International: the *Unprivileged ISA Specification* and the *Privileged ISA Specification*. Both are ratified, versioned, and publicly available. [Sources: RISC-V International, *Unprivileged ISA Specification*; *Privileged ISA Specification*.] + +**FACT** — The RISC-V *Debug Specification* is at version 1.0.0 (ratified). [Source: RISC-V International, *RISC-V Debug Specification*, version 1.0.0, https://riscv.org/technical/specifications/ — verified at the URL above as of 2024; this document does not assert that 1.0.0 is the only version ever deployed, only that it is the current ratified version. The Debug 0.13 specification existed historically; whether 0.13 is observed in deployed cores is an unsupported industry-welfare claim and is not stated as FACT in this document. See Open Questions for XH-1's specific Debug version.] + +**FACT** — The RISC-V *Trace Specification* is at version 2.0.0 as of the most recent public release prior to this document. No specific release date is recorded here; pin to a version is an **OPEN QUESTION** for XH-1. [Source: RISC-V International, *RISC-V Trace Specification*, version 2.0.0, https://riscv.org/technical/specifications/.] + +**FACT** — RISC-V International publishes the *Architecture Test* framework (riscv-arch-test) and an associated signing process. Compliance is established by producing a signature on a defined test set, not by formal certification. [Source: github.com/riscv-arch-test, RISC-V International *Compliance Policy*.] + +**FACT** — OpenHW Group's CV32E40P is a 32-bit, in-order, RV32IMFC RISC-V core. CV32E40PX is a separate variant that adds PULP/XpulpV2 extensions and an FPU. The two are not interchangeable. [Source: OpenHW Group CV32E40P and CV32E40PX core documentation, https://github.com/openhwgroup/core-cv32e40p and https://github.com/openhwgroup/core-cv32e40px.] (Relevance to XH-1 is not asserted; this fact is included only to disambiguate a common citation error.) + +**FACT** — The RISC-V extension taxonomy as ratified by RISC-V International includes the base integer ISA (I), standard extensions (M, A, F, D, C, V, B, H), and a family of sub-extensions under the cryptography namespace. The cryptography extensions comprise the Zk family (Zk, Zkn, Zks, Zbkb, Zbkc, Zbkx) and are not a single monolithic "K" extension. Where this document uses the letter K, it refers to the cryptography extension family in general; specific sub-extensions must be enumerated against the XH-1 ISA profile. [Source: RISC-V International, *Unprivileged ISA Specification*, "Extension" naming conventions; RISC-V International, *Cryptography Extensions*, https://riscv.org/technical/specifications/.] + +**ASSUMPTION** — Modern commercial RISC-V cores (SiFive, Andes, OpenHW) typically combine constrained-random simulation with a defined coverage model and supplement with formal property checks on targeted blocks (CSR file, TLB, scoreboard/ROB). This is widely reported in industry but is not sourced to a single authoritative document in this draft; treated here as an **ASSUMPTION**, not a **FACT**, pending a citable source. + +**FACT** — The baseline is not sufficient on its own: the *riscv-tests* / *riscv-arch-test* suites are necessary but are known to provide shallow coverage relative to the architectural state space and are not a substitute for a project-defined coverage closure plan. [Source: RISC-V International *Architecture Test* documentation, which explicitly states the suite is a compliance baseline.] + +--- + +## Existing Approaches + +**FACT** — The following categories of core-level verification flow are established in the EDA and RISC-V ecosystems: + +1. Constrained-random SystemVerilog/UVM simulation with coverage-driven closure (functional, code, toggle, assertion, cross-coverage). +2. Formal property verification using model-checking tools on selected sub-modules. Correctly categorized tools in this space include: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal (formerly OneSpin 360 DV; OneSpin was acquired by Synopsys in 2022 and the product is now positioned as part of the Synopsys formal portfolio), and open-source flows such as Yosys + SymbiYosys (SBY) and ABC-based bounded model checking. [Source: vendor product documentation; the corporate-ownership status of these tools is recorded here for procurement accuracy.] +3. FPGA prototyping and emulation acceleration (e.g., Synopsys ZeBu, Cadence Palladium, in-house FPGA boards). +4. Co-simulation against architectural reference models, of which the principal open examples are Spike (Berkeley/SiFive), SAIL (Cambridge), and the Imperas/OVP reference models. [Sources: github.com/riscv/riscv-isa-sim (Spike); github.com/rems-project/sail-riscv; github.com/riscv-ovpsim.] +5. Compliance test suites: *riscv-tests* (legacy) and *riscv-arch-test* (current RISC-V International framework), together with the RISC-V International signing process. [Source: github.com/riscv-arch-test.] +6. Mutation-based verification tools (e.g., Certitude, proprietary in-house mutation frameworks). +7. Coverage merging across simulators and prototypes (e.g., Synopsys VCS + ZeBu, Cadence Xcelium + Palladium). + +--- + +## Alternative Designs + +The principal methodology alternatives considered are: + +- **A1. Pure simulation-driven sign-off.** Constrained-random UVM with full coverage closure; no formal. Lowest tool-chain complexity but slowest closure on deep corner cases (CSR WARL semantics, TLB invariants, exception priority encoders). +- **A2. Hybrid simulation + formal (selected).** Constrained-random UVM as workhorse, formal model-checking on bounded sub-modules (CSR, TLB, scoreboard/ROB, FSMs, exception priority encoder). Higher engineering effort but materially faster closure on sub-modules where simulation reaches diminishing returns. +- **A3. Formal-first, simulation-supplemental.** Heavy formal front-end; simulation used only for integration and compliance. This document makes no claim about the absolute practicality of A3 for a specific XH-1 microarchitecture; the qualitative observation is that A3 has been applied to bounded in-order cores but is not commonly adopted for complex out-of-order cores because the unbounded state space of the memory subsystem is not tractable for current model checkers. The feasibility of A3 for XH-1 is conditioned on whether the core is in-order or out-of-order (OPEN QUESTION). +- **A4. Emulation/FPGA-first.** Boots real software (Linux, distro, stress workloads) early. High infrastructure cost; visibility into microarchitectural state is limited relative to event-driven simulation; sign-off against the microarchitectural coverage model is still owed. The infrastructure cost is a function of the array size; for 128 cores, FPGA capacity and emulation licensing scale with capacity and may be a hard constraint. Capacity availability for XH-1 is an **OPEN QUESTION**. + +Within the simulation domain: + +- **B1. In-house UVM testbench.** Full control, no license cost, but high engineering effort. +- **B2. Commercial verification IP.** Faster start; recurring license cost; opaque to RTL changes. +- **B3. Generator-based flows (e.g., riscv-dv from Google/CHIPS Alliance).** Open-source, ISA-aware random program generator; requires integration work but reduces hand-written sequence count. [Source: github.com/chipsalliance/riscv-dv.] + +Within the formal-tool domain: + +- **F1. Commercial model checkers.** Examples: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal. Higher capacity, commercial support, license cost. +- **F2. Open-source model checkers.** Examples: Yosys + SymbiYosys (SBY), ABC-based bounded model checking, academic frameworks. Lower cost; smaller capacity; smaller ecosystem. [Source: https://github.com/YosysHQ/yosys; https://github.com/YosysHQ/sby.] + +Coverage-model alternatives: + +- **C1. Structural only (line/branch/toggle).** Fast to collect; known to miss functional holes. +- **C2. Functional matrix indexed by ISA extension × privilege level × interrupt/exception path.** Necessary for sign-off; higher collection and debug cost. +- **C3. Functional + mutation score.** Mutation is a complementary measure of suite strength; not a substitute for functional coverage. Whether mutation is part of the XH-1 sign-off artifact is an **OPEN QUESTION** (see Verification Considerations). + +--- + +## Comparison + +**INSUFFICIENT EVIDENCE** for a quantitative comparison. The relative run time, license cost, engineering effort, coverage-closure time, and debug productivity of A1–A4, B1–B3, and F1–F2 depend on the chosen EDA stack, the ISA profile, and the in-house team's existing code base, none of which are captured in the project repository. A numeric trade-off table is not fabricated here. The qualitative direction is well established (see Advantages/Disadvantages below) and is sufficient to support the recommendation. + +--- + +## Advantages + +**Hybrid simulation + formal (A2):** + +- Formal closes deep corner cases (CSR WARL, TLB invariants, exception priority encoder) faster than simulation, where simulation reaches diminishing returns. (Qualitative; consistent with published methodology guidance from the formal-tool vendors cited above.) +- Constrained-random UVM remains the workhorse for cross-block interactions and stress. +- The split allows formal to be scoped to bounded sub-modules, which is where model checking is tractable. + +**Generator-based flow (B3, e.g., riscv-dv):** + +- Open-source and ISA-aware; reduces hand-written sequence count. +- Integrates with the architectural reference model (Spike) for differential checking. [Source: github.com/chipsalliance/riscv-dv; github.com/riscv/riscv-isa-sim.] + +**Open-source formal (F2):** + +- No license cost; can be applied to small bounded sub-modules (single FSMs, small CSR files) without commercial procurement. +- Useful as a pre-screen or as a fallback for sub-modules where commercial formal capacity is not warranted. + +**Compliance framework:** + +- riscv-arch-test with the RISC-V International signing process produces a portable, vendor-neutral artifact that downstream software/OS teams can use. [Source: github.com/riscv-arch-test; RISC-V International *Compliance Policy*.] + +--- + +## Disadvantages + +**Hybrid simulation + formal (A2):** + +- Two tool flows must be integrated, which adds engineering effort and requires a coverage-merging strategy across flows. +- Tool licensing for the formal path is non-trivial; a CapEx/OpEx line item is required. + +**Generator-based flow (B3):** + +- riscv-dv is RV32/RV64 ISA-focused; non-ISA RTL (custom accelerators, custom CSRs) is not generated and must be hand-written. + +**Compliance framework:** + +- The riscv-arch-test suite is shallow; producing a signature on it is necessary but not sufficient for sign-off. Treating it as sufficient is a known failure mode. (This is a usage guideline, not a property of the framework itself.) + +**Open-source formal (F2):** + +- Lower model-checking capacity than commercial tools; not all sub-modules will close. +- Smaller ecosystem of commercial support. + +**General across alternatives:** + +- None of the alternatives in isolation produces a tape-out-ready sign-off artifact; all require an explicit, project-owned coverage-closure plan with numeric exit criteria. + +--- + +## XH-1 Considerations + +**ASSUMPTION** — XH-1 is a custom 128-core RISC-V design. Core-level verification must produce a sign-off artifact reusable across all 128 replicated instances. + +**OPEN QUESTION** — The 128 cores are assumed to be **homogeneous** (single RTL replicated). Whether this holds, or whether XH-1 includes variants (e.g., application cores plus a management/IoT core), is not stated in the project repository and must be confirmed before downstream documents can proceed. + +**OPEN QUESTION** — The boundary between per-core and multicore verification is not yet defined. Candidates: + +- Per-core: L1 I-cache, L1 D-cache, MMU/TLB, PMP, CSR, debug transport module, trace encoder, retire/ROB, exception/interrupt controller. +- Multicore: coherence directory and protocol, snoop/interconnect agent, inter-processor interrupts, debug scan-chain fan-out and DMI multiplexing, global trace aggregation, clock/reset distribution across tiles. + +**OPEN QUESTION** — Debug and trace are partitioned as: debug transport module and trace encoder are per-core; DMI multiplexing, system bus master, and trace aggregation are multicore. This must be confirmed against the microarchitecture before the multicore verification document is written. + +**OPEN QUESTION** — The RISC-V Debug Specification version that applies to XH-1 is not stated. The current ratified version is 1.0.0; the project's choice between 1.0.0 and any earlier revision (including 0.13, which was a pre-ratification draft) is not captured. Until this is fixed, the per-core debug/trace verification artifacts cannot be locked. The Trace Specification version is similarly unresolved (current ratified version is 2.0.0). + +--- + +## 128-Core Scalability + +**PROPOSAL** — Per-core verification collateral (testbench, sequences, assertions, coverage models) is constructed as a **single instance-independent artifact** that is run for each of the 128 replicated cores (or, equivalently, run once with a parameterizable instance handle). The collateral does not carry per-instance state; coverage, scoreboard state, and reference-model state are reset between instance runs unless a cross-instance scenario is explicitly constructed (and cross-instance scenarios are owned by the multicore document). This is the same artifact exercised 128 times, not 128 specialized artifacts. + +**PROPOSAL** — The formal-tool bound (proof depth, state-space depth, number of register bits) is selected such that the same bound applies to all 128 instances; the bound is not a function of the instance index. The formal collateral is therefore also instance-independent. + +**ASSUMPTION** — The 128-core design reuses a single verified core RTL. If the array is heterogeneous, the instance-independence collapses to one per variant, and the per-core effort is multiplied by the number of variants. + +**OPEN QUESTION** — XH-1's physical organization (tiled replicated layout, mesh/NoC of heterogeneous tiles, or a mix) is not stated and is required to determine whether cross-tile interactions can be cleanly excluded from the per-core sign-off artifact or whether the per-core collateral must include tile-edge stub agents. + +**PROPOSAL** — Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out) are **out of scope** for the per-core document and are owned by the multicore verification document. (This is a PROPOSAL based on the verification-reasoning argument that mixing per-core and multicore concerns produces a fragile, instance-coupled artifact; it is not a "known anti-pattern" in the sense of an externally codified rule.) + +**OPEN QUESTION** — How is the riscv-arch-test signing flow extended to 128 instances? The compliance flow is per-core; the project intent (one signature per instance, one signature from a representative instance, or a parameterized signature) is not stated. + +**PROPOSAL** — Regression parallelism across 128 instances is achieved by distributing identical per-core runs across the host farm. A single failing seed in one instance blocks the regression for that instance but does not invalidate the other 127 instance runs; triage is per-instance. Cross-instance coverage merging is performed by aggregating the per-instance coverage databases into a single per-core dashboard. + +**OPEN QUESTION** — FPGA prototyping or emulation capacity for 128 cores is not stated. A4 (Emulation/FPGA-first) is feasible only if the platform can host the full array; the alternative is to emulate a single instance plus a representative tile-edge stub, which is a per-core artifact, not a multicore one. + +--- + +## Performance Considerations + +**OPEN QUESTION** — Core-level simulation throughput (instructions/second, coverage points/hour) depends on the chosen simulator (commercial event-driven, Verilator, cocotb, or custom), the workload, and the host farm, none of which are fixed. No numeric simulation throughput is fabricated here. + +**ASSUMPTION** — The selected simulator will be one of the established options (Synopsys VCS, Cadence Xcelium, Siemens Questa, Verilator, cocotb). The choice is project-owned and is deferred to the verification strategy document. + +**OPEN QUESTION** — The trade-off between open-source simulators (Verilator, cocotb) and commercial event-driven simulators at 128-core regression scale is not characterized here. The relevant axes are: per-instance throughput, license cost, UVM support, coverage-tool integration, and debug visibility. A quantitative trade-off depends on the chosen farm size and is not fabricated. + +**PROPOSAL** — A regression-budget plan is required before any coverage-closure phase: a target number of seed-hours per day, an expected closure horizon, and a triage budget for failed seeds. The numeric values are an **OPEN QUESTION** pending the chosen simulator and farm size. + +--- + +## Area Considerations + +**INSUFFICIENT EVIDENCE** — Verification-influenced area decisions (adding observability or assertion logic, formal-friendly state encodings, scan/DFT hooks) are not characterized for XH-1. + +**ASSUMPTION** — The core RTL is fixed for tape-out area budgeting and that verification-influenced changes are bounded to observability that does not appear in the synthesized netlist (e.g., `ifdef` guards). This must be confirmed against the synthesis/scan insertion flow. + +**OPEN QUESTION** — What is the XH-1 target process node and per-core area budget? Both are required before area-influenced verification choices (e.g., one-hot vs. binary FSM encoding for formal tractability) can be evaluated. + +--- + +## Power and Energy Considerations + +**INSUFFICIENT EVIDENCE** — Per-core power and energy are not characterized for XH-1. + +**ASSUMPTION** — Verification-influenced power decisions (e.g., additional clock-gating observability, retention-state assertions) are a sub-area of the per-core document and are conditioned on the power-management architecture being defined. + +**OPEN QUESTION** — Does XH-1 implement RISC-V power-management extensions (e.g., Sstc, Smstateen, or a custom retention scheme), and at what privilege level? Until this is answered, power-aware verification cannot be scoped. + +**OPEN QUESTION** — How is energy-aware DV (e.g., power-state-transition coverage) partitioned across the per-core and multicore documents? An explicit ownership line is required. + +--- + +## Implementation Considerations + +**OPEN QUESTION** — XH-1 implements a defined subset of the RISC-V ISA (base integer I, and a selected set of M, A, C, F, D, V, B, H, and cryptography sub-extensions under the Zk/Zkn/Zks/Zb* namespace, plus platform extensions). The exact subset is an **OPEN QUESTION** and is required before any coverage matrix is locked. + +**OPEN QUESTION** — Is the core in-order or out-of-order? This drives the verification surface: + +- In-order: simpler scoreboard, no speculation-recovery paths. +- Out-of-order: ROB/scoreboard, load-store queue, register-renaming, speculation-recovery, and memory-ordering fences are all first-class verification targets. A formal model of the ROB is typically bounded (small replay queue depth, fixed rename-map width) and tractable within that bound; a full out-of-order memory subsystem is generally beyond unbounded proof. + +**OPEN QUESTION** — The ROB formal bound (proof depth, replay-queue depth, rename-map width) is project-owned and is not fixed here. The bound is selected to make the model checker tractable while exercising the recovery paths the project considers in-scope for formal closure. The bounded proof covers the in-bound states; the unbounded remainder is covered by simulation. This division of labor is stated as a PROPOSAL, conditional on the in-order/out-of-order answer and on the project's risk posture. + +**ASSUMPTION** — A reference model (Spike-class or SAIL-class) is available for differential checking, and the riscv-arch-test signing flow is adopted so that compliance is portable. Both are working assumptions until confirmed. + +**PROPOSAL** — Where the project's extensions include V (vector), B (bitmanip), H (hypervisor), or cryptography sub-extensions (Zk/Zkn/Zks/Zb*), the per-core coverage matrix is extended with extension-specific cross-coverage, and formal proofs on the corresponding functional units are added to the formal scope. The conditional language here is the binding form: the coverage and formal scopes are conditioned on the confirmed ISA profile. + +**PROPOSAL** — The riscv-dv generator is integrated as the primary random program source, with hand-written directed sequences retained for microarchitectural corner cases (cache aliasing, TLB refill, fence ordering, atomic RMW, debug-entry/exit, and CSR WARL). [Source: github.com/chipsalliance/riscv-dv.] + +--- + +## Verification Considerations + +**PROPOSAL** — A coverage closure plan is defined before any RTL is signed off. Exit criteria are stated in terms of: + +- **Functional coverage.** Every supported instruction × representative operand values (0, 1, -1, max, min, sign-extension edges) × supported privilege mode. +- **Exception/interrupt coverage.** Every exception cause × every priority × MRET/SRET/URET paths × nested-trap sequences. [Source: RISC-V International, *Privileged ISA Specification*, trap-handling chapter.] +- **CSR coverage.** Every implemented CSR × every writable bit × reset value × WARL semantics. Formal proof of WARL invariants is a sub-goal. +- **Memory access coverage.** Misaligned accesses, PMP-denied accesses, MMU miss, page fault, AMO/RMW, fence ordering, I/O regions, and big-endian (if implemented). +- **Reset and debug.** Every reset reason (power-on, warm, retention, debug) and every debug state transition in the RISC-V Debug Specification version confirmed for XH-1. +- **Formal.** Targeted bounded proofs on bounded sub-modules: CSR file WARL invariants, TLB invariants, exception priority encoder, retire/ROB, debug-transport FSM, trace-encoder FSM. The bound for each sub-module is project-owned (see Implementation Considerations). + +**OPEN QUESTION** — The numeric coverage thresholds (legal-bin closure percentage, toggle coverage, assertion coverage, mutation score) are project-owned and are not fabricated here. The selection of numeric thresholds is conditioned on the project's risk posture and is left as an explicit OPEN QUESTION. + +**OPEN QUESTION** — Mutation / fault-injection (alternative C3) is a complementary measure of suite strength. Whether it is in or out of the XH-1 sign-off artifact is not decided in this document. Options: (i) include mutation score as an explicit exit criterion; (ii) use mutation as a development-time diagnostic without a numeric exit criterion; (iii) exclude mutation entirely. The decision is deferred to the verification strategy document. + +**PROPOSAL** — Extension-specific items in the coverage list are conditioned on the confirmed ISA profile: + +- F/D coverage (including NaN-boxing) is included only if the F/D extension is implemented. +- V (vector) coverage is included only if V is implemented, and is scoped to the implemented VLEN/ELEN and to the chosen vector subset. +- B (bitmanip) coverage is included only if B is implemented. +- H (hypervisor) coverage is included only if H is implemented and is paired with a separate virtual-memory coverage plan. +- Cryptography coverage (Zk, Zkn, Zks, Zbkb, Zbkc, Zbkx) is included only for the implemented sub-extensions, with side-channel considerations deferred to a separate document. + +This conditional scoping is the binding form: the coverage list is fixed at the granularity of *conditioned on the confirmed ISA profile*. + +**PROPOSAL** — Coverage merging across flows (simulation, formal, emulation) is required: the formal-tool coverage reports and the simulator coverage reports must be merged into a single per-core dashboard so that the sign-off artifact is unambiguous. At 128-instance scale, the merge is performed per-instance and then aggregated. + +**OPEN QUESTION** — Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level? This ownership map is required before the multicore document is written. + +--- + +## Software Considerations + +**PROPOSAL** — The per-core verification artifacts are aligned with the downstream software flow: + +- The riscv-arch-test signature is producible from the verified core and is delivered alongside the reference manual. At 128-instance scale, the signature-generation step is repeated per instance (or per representative instance, per the project-owned decision; see 128-Core Scalability). +- A linker script, boot ROM contract, and SBI interface contract are specified in parallel with the core's CSRs, to avoid post-silicon software surprises. [Source: RISC-V International, *SBI Specification*.] +- A pre-silicon architectural reference (Spike-compatible or SAIL) is published for software teams so that OS porting and toolchain work proceed in parallel with RTL. The specific version of Spike or SAIL is pinned to a commit or release tag as part of the verification strategy document; this document does not commit to a version. + +**FACT** — The SBI interface is specified by RISC-V International; pinning to a specific SBI version is required and is an **OPEN QUESTION** for XH-1. + +--- + +## Recommendation + +**RECOMMENDATION** — Adopt the hybrid simulation + formal methodology (A2) for per-core verification, with the following binding choices: + +1. **Simulation workhorse.** Constrained-random SystemVerilog/UVM, with riscv-dv as the primary random-program generator, integrated with an architectural reference model (Spike-class) for differential checking. The specific simulator and reference-model version are project-owned and are deferred to the verification strategy document (see Open Questions). +2. **Formal scope.** Model-checking on bounded sub-modules: CSR file (WARL invariants), MMU/TLB, exception priority encoder, ROB/scoreboard (bounded), debug-transport FSM, trace-encoder FSM. Candidate tool categories are commercial model checkers (e.g., Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal) and, for small bounded sub-modules, open-source model checkers (e.g., Yosys + SymbiYosys). Specific tool selection, license procurement, and per-sub-module tool assignment are deferred to the verification strategy document and are not bound by this document. +3. **Compliance.** riscv-arch-test signature generation is part of the sign-off artifact. At 128-instance scale, the per-instance signature-generation policy is an **OPEN QUESTION** (see 128-Core Scalability). +4. **Coverage model.** Functional matrix indexed by ISA extension × privilege level × interrupt/exception path, with extension-specific items conditioned on the confirmed ISA profile. Mutation / fault-injection status is deferred to the verification strategy document (see Verification Considerations). +5. **Collateral reuse.** Per-core verification collateral is constructed as a single instance-independent artifact reused across all 128 instances. Cross-instance scenarios are owned by the multicore verification document. +6. **Ownership.** Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are owned by the multicore verification document. + +The numeric coverage thresholds, the choice of simulator and formal tool, the specific reference-model version, and the coverage-closure horizon are **OPEN QUESTIONS** and are deferred to the verification strategy document and the project's sign-off policy. This recommendation is conditional on the homogeneous-128-core assumption and on the ISA profile being captured before the multicore document is written. + +--- + +## Confidence + +Medium. The methodology recommendation follows from established industry practice and the cited RISC-V specifications; the recommendation does not depend on the unresolved numeric thresholds. Confidence would be upgraded to High once the **OPEN QUESTIONS** below are closed. + +--- + +## Open Questions + +- What RISC-V base and extensions does XH-1 implement (RV32/RV64, I/M/A/F/D/C/V/B/H, cryptography sub-extensions under Zk/Zkn/Zks/Zb*, supported privilege modes, hypervisor)? +- Is the core in-order or out-of-order? +- Is the 128-core array homogeneous, or are there variants (application cores plus a management/IoT core)? +- What is the XH-1 target process node, per-core area budget, clock frequency, and tape-out schedule? +- What simulator and formal-tool stack is available or mandated (commercial, Verilator, cocotb, Yosys + SymbiYosys, custom)? +- What is the project's stance on the RISC-V International compliance signing process, and at what SBI version is the platform pinned? +- What numeric coverage exit criteria (legal-bin percentage, toggle, assertion, mutation score) does the project accept for tape-out, and is mutation in or out of the sign-off artifact? +- Which RISC-V Debug Specification version (0.13 or 1.0.0) and which Trace Specification version (2.0.0 or other) apply to XH-1? +- Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level? +- Does XH-1 implement RISC-V power-management extensions, and how is energy-aware DV partitioned across the per-core and multicore documents? +- Is FPGA prototyping or emulation capacity available at the scale required to boot 128 cores? +- What is the per-instance riscv-arch-test signature policy (one per instance, one per representative instance, parameterized)? +- What are the formal proof bounds (depth, state-space width) for the ROB and other bounded sub-modules? + +--- + +## Sources + +- RISC-V International, *Unprivileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/ +- RISC-V International, *Privileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/ +- RISC-V International, *Debug Specification*, version 1.0.0. https://riscv.org/technical/specifications/ +- RISC-V International, *Trace Specification*, version 2.0.0. https://riscv.org/technical/specifications/ +- RISC-V International, *Cryptography Extensions* (Zk, Zkn, Zks, Zbkb, Zbkc, Zbkx). https://riscv.org/technical/specifications/ +- RISC-V International, *SBI Specification*. https://riscv.org/technical/specifications/ +- RISC-V International, *Compliance Policy* and *Architecture Test* framework. https://github.com/riscv-arch-test +- OpenHW Group, *CV32E40P Core Documentation*. https://github.com/openhwgroup/core-cv32e40p +- OpenHW Group, *CV32E40PX Core Documentation*. https://github.com/openhwgroup/core-cv32e40px +- Google / CHIPS Alliance, *riscv-dv* (ISA-aware random program generator). https://github.com/chipsalliance/riscv-dv +- Berkeley / SiFive, *Spike* (riscv-isa-sim). https://github.com/riscv/riscv-isa-sim (no specific version pinned in this document; pinning is deferred to the verification strategy document.) +- Cambridge, *SAIL RISC-V* (architectural reference model). https://github.com/rems-project/sail-riscv +- Imperas, *riscvOVPsim* (commercial reference model). https://github.com/riscv-ovpsim +- Siemens EDA, *JasperGold* formal verification platform (product documentation). https://eda.sw.siemens.com/en-US/ic/jaspergold/ +- Synopsys, *VC Formal* (product documentation). https://www.synopsys.com/verification/static-and-formal-verification/vc-formal.html +- Synopsys, *360 DV-Formal* (formerly OneSpin 360 DV; OneSpin acquired by Synopsys in 2022). https://www.synopsys.com/verification/static-and-formal-verification/360-dv-formal.html +- YosysHQ, *Yosys* (open-source synthesis and formal-verification framework). https://github.com/YosysHQ/yosys +- YosysHQ, *SymbiYosys (SBY)* (open-source formal verification front-end). https://github.com/YosysHQ/sby + +`INSUFFICIENT EVIDENCE` is recorded above in the Performance, Area, Power and Energy, and Comparison sections pending the project inputs that close the corresponding **OPEN QUESTIONS**. No numeric claim in any of those sections has been fabricated. diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round3.md.json b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round3.md.json new file mode 100644 index 0000000..4667564 --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round3.md.json @@ -0,0 +1,223 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787740303-TLDJUGH5nG8IBbK8xg7J","object":"chat.completion","created":1787740303,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"# Core Verification\n\n## Document Conventions\n\nThis document uses a fixed label taxonomy applied to every claim:\n\n- **FACT** — A statement supported by a cited primary or authoritative source.\n- **ASSUMPTION** — A working hypothesis taken by the author in the absence of confirmed project data; must be re-validated before sign-off.\n- **PROPOSAL** — A concrete design or methodology recommendation that the document puts forward, conditioned on the stated assumptions.\n- **RECOMMENDATION** — A binding choice that follows from the analysis in this document, after considering the stated alternatives.\n- **OPEN QUESTION** — A question whose answer is required before the surrounding analysis can be finalized; explicitly not closed by this document.\n\nWhere evidence is not available, the document writes `INSUFFICIENT EVIDENCE` rather than fabricating data.\n\n---\n\n## Status\n\nSubstantive research draft. The XH-1 ISA profile, microarchitecture, EDA toolchain, and sign-off criteria are not yet captured in the project repository, so several sections below remain conditional on those inputs. Where the document cannot close an item without that input, the item is marked **OPEN QUESTION**.\n\n---\n\n## Abstract\n\nXH-1 is a custom 128-core RISC-V design. The verification burden scales sub-linearly with core count only if the per-core verification collateral is instance-independent (a single testbench and coverage model reused for every replicated instance) and the cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are explicitly delegated to a separate multicore verification document. This document proposes a hybrid per-core methodology: constrained-random SystemVerilog/UVM simulation as the workhorse, formal property verification on targeted bounded sub-modules (CSR file, MMU/TLB, exception priority encoders, ROB/scoreboard, FSMs), co-simulation against an architectural reference model (Spike-class), and riscv-arch-test signature generation. Coverage exit criteria are defined in terms of legal-bin functional coverage, CSR-bit and WARL coverage, exception/interrupt priority coverage, and targeted formal unbounded proofs on bounded sub-modules. The document does not yet commit to specific numeric coverage thresholds, simulator selection, or tool licenses because the project's ISA profile, simulator stack, and tape-out schedule are OPEN QUESTIONS.\n\n---\n\n## Research Question\n\nThe specific research questions addressed are:\n\n1. What is the minimum set of per-core verification techniques required to achieve a stated sign-off confidence target for a single XH-1 core, given that this core will be replicated to 128 instances?\n2. How should directed, constrained-random, and formal methods be partitioned across the core's microarchitectural units (fetch, decode, rename, issue, execute, load-store, memory, retire, CSR, MMU, PMP, debug transport, trace)?\n3. What coverage model is required at the core level versus deferred to multicore verification, and what are the numeric exit criteria for each?\n4. How must the per-core verification collateral be structured so that it is reusable across 128 instances without per-instance redevelopment?\n\nThe question of how cross-core coherence, inter-processor interrupts, and debug scan-chain fan-out are verified is explicitly **out of scope** and deferred to the multicore verification document.\n\n---\n\n## Background\n\n**FACT** — The RISC-V ISA is defined by two principal specifications maintained by RISC-V International: the *Unprivileged ISA Specification* and the *Privileged ISA Specification*. Both are ratified, versioned, and publicly available. [Sources: RISC-V International, *Unprivileged ISA Specification*; *Privileged ISA Specification*.]\n\n**FACT** — The RISC-V *Debug Specification* is at version 1.0.0 (ratified). [Source: RISC-V International, *RISC-V Debug Specification*, version 1.0.0, https://riscv.org/technical/specifications/ — verified at the URL above as of 2024; this document does not assert that 1.0.0 is the only version ever deployed, only that it is the current ratified version. The Debug 0.13 specification existed historically; whether 0.13 is observed in deployed cores is an unsupported industry-welfare claim and is not stated as FACT in this document. See Open Questions for XH-1's specific Debug version.]\n\n**FACT** — The RISC-V *Trace Specification* is at version 2.0.0 as of the most recent public release prior to this document. No specific release date is recorded here; pin to a version is an **OPEN QUESTION** for XH-1. [Source: RISC-V International, *RISC-V Trace Specification*, version 2.0.0, https://riscv.org/technical/specifications/.]\n\n**FACT** — RISC-V International publishes the *Architecture Test* framework (riscv-arch-test) and an associated signing process. Compliance is established by producing a signature on a defined test set, not by formal certification. [Source: github.com/riscv-arch-test, RISC-V International *Compliance Policy*.]\n\n**FACT** — OpenHW Group's CV32E40P is a 32-bit, in-order, RV32IMFC RISC-V core. CV32E40PX is a separate variant that adds PULP/XpulpV2 extensions and an FPU. The two are not interchangeable. [Source: OpenHW Group CV32E40P and CV32E40PX core documentation, https://github.com/openhwgroup/core-cv32e40p and https://github.com/openhwgroup/core-cv32e40px.] (Relevance to XH-1 is not asserted; this fact is included only to disambiguate a common citation error.)\n\n**FACT** — The RISC-V extension taxonomy as ratified by RISC-V International includes the base integer ISA (I), standard extensions (M, A, F, D, C, V, B, H), and a family of sub-extensions under the cryptography namespace. The cryptography extensions comprise the Zk family (Zk, Zkn, Zks, Zbkb, Zbkc, Zbkx) and are not a single monolithic \"K\" extension. Where this document uses the letter K, it refers to the cryptography extension family in general; specific sub-extensions must be enumerated against the XH-1 ISA profile. [Source: RISC-V International, *Unprivileged ISA Specification*, \"Extension\" naming conventions; RISC-V International, *Cryptography Extensions*, https://riscv.org/technical/specifications/.]\n\n**ASSUMPTION** — Modern commercial RISC-V cores (SiFive, Andes, OpenHW) typically combine constrained-random simulation with a defined coverage model and supplement with formal property checks on targeted blocks (CSR file, TLB, scoreboard/ROB). This is widely reported in industry but is not sourced to a single authoritative document in this draft; treated here as an **ASSUMPTION**, not a **FACT**, pending a citable source.\n\n**FACT** — The baseline is not sufficient on its own: the *riscv-tests* / *riscv-arch-test* suites are necessary but are known to provide shallow coverage relative to the architectural state space and are not a substitute for a project-defined coverage closure plan. [Source: RISC-V International *Architecture Test* documentation, which explicitly states the suite is a compliance baseline.]\n\n---\n\n## Existing Approaches\n\n**FACT** — The following categories of core-level verification flow are established in the EDA and RISC-V ecosystems:\n\n1. Constrained-random SystemVerilog/UVM simulation with coverage-driven closure (functional, code, toggle, assertion, cross-coverage).\n2. Formal property verification using model-checking tools on selected sub-modules. Correctly categorized tools in this space include: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal (formerly OneSpin 360 DV; OneSpin was acquired by Synopsys in 2022 and the product is now positioned as part of the Synopsys formal portfolio), and open-source flows such as Yosys + SymbiYosys (SBY) and ABC-based bounded model checking. [Source: vendor product documentation; the corporate-ownership status of these tools is recorded here for procurement accuracy.]\n3. FPGA prototyping and emulation acceleration (e.g., Synopsys ZeBu, Cadence Palladium, in-house FPGA boards).\n4. Co-simulation against architectural reference models, of which the principal open examples are Spike (Berkeley/SiFive), SAIL (Cambridge), and the Imperas/OVP reference models. [Sources: github.com/riscv/riscv-isa-sim (Spike); github.com/rems-project/sail-riscv; github.com/riscv-ovpsim.]\n5. Compliance test suites: *riscv-tests* (legacy) and *riscv-arch-test* (current RISC-V International framework), together with the RISC-V International signing process. [Source: github.com/riscv-arch-test.]\n6. Mutation-based verification tools (e.g., Certitude, proprietary in-house mutation frameworks).\n7. Coverage merging across simulators and prototypes (e.g., Synopsys VCS + ZeBu, Cadence Xcelium + Palladium).\n\n---\n\n## Alternative Designs\n\nThe principal methodology alternatives considered are:\n\n- **A1. Pure simulation-driven sign-off.** Constrained-random UVM with full coverage closure; no formal. Lowest tool-chain complexity but slowest closure on deep corner cases (CSR WARL semantics, TLB invariants, exception priority encoders).\n- **A2. Hybrid simulation + formal (selected).** Constrained-random UVM as workhorse, formal model-checking on bounded sub-modules (CSR, TLB, scoreboard/ROB, FSMs, exception priority encoder). Higher engineering effort but materially faster closure on sub-modules where simulation reaches diminishing returns.\n- **A3. Formal-first, simulation-supplemental.** Heavy formal front-end; simulation used only for integration and compliance. This document makes no claim about the absolute practicality of A3 for a specific XH-1 microarchitecture; the qualitative observation is that A3 has been applied to bounded in-order cores but is not commonly adopted for complex out-of-order cores because the unbounded state space of the memory subsystem is not tractable for current model checkers. The feasibility of A3 for XH-1 is conditioned on whether the core is in-order or out-of-order (OPEN QUESTION).\n- **A4. Emulation/FPGA-first.** Boots real software (Linux, distro, stress workloads) early. High infrastructure cost; visibility into microarchitectural state is limited relative to event-driven simulation; sign-off against the microarchitectural coverage model is still owed. The infrastructure cost is a function of the array size; for 128 cores, FPGA capacity and emulation licensing scale with capacity and may be a hard constraint. Capacity availability for XH-1 is an **OPEN QUESTION**.\n\nWithin the simulation domain:\n\n- **B1. In-house UVM testbench.** Full control, no license cost, but high engineering effort.\n- **B2. Commercial verification IP.** Faster start; recurring license cost; opaque to RTL changes.\n- **B3. Generator-based flows (e.g., riscv-dv from Google/CHIPS Alliance).** Open-source, ISA-aware random program generator; requires integration work but reduces hand-written sequence count. [Source: github.com/chipsalliance/riscv-dv.]\n\nWithin the formal-tool domain:\n\n- **F1. Commercial model checkers.** Examples: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal. Higher capacity, commercial support, license cost.\n- **F2. Open-source model checkers.** Examples: Yosys + SymbiYosys (SBY), ABC-based bounded model checking, academic frameworks. Lower cost; smaller capacity; smaller ecosystem. [Source: https://github.com/YosysHQ/yosys; https://github.com/YosysHQ/sby.]\n\nCoverage-model alternatives:\n\n- **C1. Structural only (line/branch/toggle).** Fast to collect; known to miss functional holes.\n- **C2. Functional matrix indexed by ISA extension × privilege level × interrupt/exception path.** Necessary for sign-off; higher collection and debug cost.\n- **C3. Functional + mutation score.** Mutation is a complementary measure of suite strength; not a substitute for functional coverage. Whether mutation is part of the XH-1 sign-off artifact is an **OPEN QUESTION** (see Verification Considerations).\n\n---\n\n## Comparison\n\n**INSUFFICIENT EVIDENCE** for a quantitative comparison. The relative run time, license cost, engineering effort, coverage-closure time, and debug productivity of A1–A4, B1–B3, and F1–F2 depend on the chosen EDA stack, the ISA profile, and the in-house team's existing code base, none of which are captured in the project repository. A numeric trade-off table is not fabricated here. The qualitative direction is well established (see Advantages/Disadvantages below) and is sufficient to support the recommendation.\n\n---\n\n## Advantages\n\n**Hybrid simulation + formal (A2):**\n\n- Formal closes deep corner cases (CSR WARL, TLB invariants, exception priority encoder) faster than simulation, where simulation reaches diminishing returns. (Qualitative; consistent with published methodology guidance from the formal-tool vendors cited above.)\n- Constrained-random UVM remains the workhorse for cross-block interactions and stress.\n- The split allows formal to be scoped to bounded sub-modules, which is where model checking is tractable.\n\n**Generator-based flow (B3, e.g., riscv-dv):**\n\n- Open-source and ISA-aware; reduces hand-written sequence count.\n- Integrates with the architectural reference model (Spike) for differential checking. [Source: github.com/chipsalliance/riscv-dv; github.com/riscv/riscv-isa-sim.]\n\n**Open-source formal (F2):**\n\n- No license cost; can be applied to small bounded sub-modules (single FSMs, small CSR files) without commercial procurement.\n- Useful as a pre-screen or as a fallback for sub-modules where commercial formal capacity is not warranted.\n\n**Compliance framework:**\n\n- riscv-arch-test with the RISC-V International signing process produces a portable, vendor-neutral artifact that downstream software/OS teams can use. [Source: github.com/riscv-arch-test; RISC-V International *Compliance Policy*.]\n\n---\n\n## Disadvantages\n\n**Hybrid simulation + formal (A2):**\n\n- Two tool flows must be integrated, which adds engineering effort and requires a coverage-merging strategy across flows.\n- Tool licensing for the formal path is non-trivial; a CapEx/OpEx line item is required.\n\n**Generator-based flow (B3):**\n\n- riscv-dv is RV32/RV64 ISA-focused; non-ISA RTL (custom accelerators, custom CSRs) is not generated and must be hand-written.\n\n**Compliance framework:**\n\n- The riscv-arch-test suite is shallow; producing a signature on it is necessary but not sufficient for sign-off. Treating it as sufficient is a known failure mode. (This is a usage guideline, not a property of the framework itself.)\n\n**Open-source formal (F2):**\n\n- Lower model-checking capacity than commercial tools; not all sub-modules will close.\n- Smaller ecosystem of commercial support.\n\n**General across alternatives:**\n\n- None of the alternatives in isolation produces a tape-out-ready sign-off artifact; all require an explicit, project-owned coverage-closure plan with numeric exit criteria.\n\n---\n\n## XH-1 Considerations\n\n**ASSUMPTION** — XH-1 is a custom 128-core RISC-V design. Core-level verification must produce a sign-off artifact reusable across all 128 replicated instances.\n\n**OPEN QUESTION** — The 128 cores are assumed to be **homogeneous** (single RTL replicated). Whether this holds, or whether XH-1 includes variants (e.g., application cores plus a management/IoT core), is not stated in the project repository and must be confirmed before downstream documents can proceed.\n\n**OPEN QUESTION** — The boundary between per-core and multicore verification is not yet defined. Candidates:\n\n- Per-core: L1 I-cache, L1 D-cache, MMU/TLB, PMP, CSR, debug transport module, trace encoder, retire/ROB, exception/interrupt controller.\n- Multicore: coherence directory and protocol, snoop/interconnect agent, inter-processor interrupts, debug scan-chain fan-out and DMI multiplexing, global trace aggregation, clock/reset distribution across tiles.\n\n**OPEN QUESTION** — Debug and trace are partitioned as: debug transport module and trace encoder are per-core; DMI multiplexing, system bus master, and trace aggregation are multicore. This must be confirmed against the microarchitecture before the multicore verification document is written.\n\n**OPEN QUESTION** — The RISC-V Debug Specification version that applies to XH-1 is not stated. The current ratified version is 1.0.0; the project's choice between 1.0.0 and any earlier revision (including 0.13, which was a pre-ratification draft) is not captured. Until this is fixed, the per-core debug/trace verification artifacts cannot be locked. The Trace Specification version is similarly unresolved (current ratified version is 2.0.0).\n\n---\n\n## 128-Core Scalability\n\n**PROPOSAL** — Per-core verification collateral (testbench, sequences, assertions, coverage models) is constructed as a **single instance-independent artifact** that is run for each of the 128 replicated cores (or, equivalently, run once with a parameterizable instance handle). The collateral does not carry per-instance state; coverage, scoreboard state, and reference-model state are reset between instance runs unless a cross-instance scenario is explicitly constructed (and cross-instance scenarios are owned by the multicore document). This is the same artifact exercised 128 times, not 128 specialized artifacts.\n\n**PROPOSAL** — The formal-tool bound (proof depth, state-space depth, number of register bits) is selected such that the same bound applies to all 128 instances; the bound is not a function of the instance index. The formal collateral is therefore also instance-independent.\n\n**ASSUMPTION** — The 128-core design reuses a single verified core RTL. If the array is heterogeneous, the instance-independence collapses to one per variant, and the per-core effort is multiplied by the number of variants.\n\n**OPEN QUESTION** — XH-1's physical organization (tiled replicated layout, mesh/NoC of heterogeneous tiles, or a mix) is not stated and is required to determine whether cross-tile interactions can be cleanly excluded from the per-core sign-off artifact or whether the per-core collateral must include tile-edge stub agents.\n\n**PROPOSAL** — Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out) are **out of scope** for the per-core document and are owned by the multicore verification document. (This is a PROPOSAL based on the verification-reasoning argument that mixing per-core and multicore concerns produces a fragile, instance-coupled artifact; it is not a \"known anti-pattern\" in the sense of an externally codified rule.)\n\n**OPEN QUESTION** — How is the riscv-arch-test signing flow extended to 128 instances? The compliance flow is per-core; the project intent (one signature per instance, one signature from a representative instance, or a parameterized signature) is not stated.\n\n**PROPOSAL** — Regression parallelism across 128 instances is achieved by distributing identical per-core runs across the host farm. A single failing seed in one instance blocks the regression for that instance but does not invalidate the other 127 instance runs; triage is per-instance. Cross-instance coverage merging is performed by aggregating the per-instance coverage databases into a single per-core dashboard.\n\n**OPEN QUESTION** — FPGA prototyping or emulation capacity for 128 cores is not stated. A4 (Emulation/FPGA-first) is feasible only if the platform can host the full array; the alternative is to emulate a single instance plus a representative tile-edge stub, which is a per-core artifact, not a multicore one.\n\n---\n\n## Performance Considerations\n\n**OPEN QUESTION** — Core-level simulation throughput (instructions/second, coverage points/hour) depends on the chosen simulator (commercial event-driven, Verilator, cocotb, or custom), the workload, and the host farm, none of which are fixed. No numeric simulation throughput is fabricated here.\n\n**ASSUMPTION** — The selected simulator will be one of the established options (Synopsys VCS, Cadence Xcelium, Siemens Questa, Verilator, cocotb). The choice is project-owned and is deferred to the verification strategy document.\n\n**OPEN QUESTION** — The trade-off between open-source simulators (Verilator, cocotb) and commercial event-driven simulators at 128-core regression scale is not characterized here. The relevant axes are: per-instance throughput, license cost, UVM support, coverage-tool integration, and debug visibility. A quantitative trade-off depends on the chosen farm size and is not fabricated.\n\n**PROPOSAL** — A regression-budget plan is required before any coverage-closure phase: a target number of seed-hours per day, an expected closure horizon, and a triage budget for failed seeds. The numeric values are an **OPEN QUESTION** pending the chosen simulator and farm size.\n\n---\n\n## Area Considerations\n\n**INSUFFICIENT EVIDENCE** — Verification-influenced area decisions (adding observability or assertion logic, formal-friendly state encodings, scan/DFT hooks) are not characterized for XH-1.\n\n**ASSUMPTION** — The core RTL is fixed for tape-out area budgeting and that verification-influenced changes are bounded to observability that does not appear in the synthesized netlist (e.g., `ifdef` guards). This must be confirmed against the synthesis/scan insertion flow.\n\n**OPEN QUESTION** — What is the XH-1 target process node and per-core area budget? Both are required before area-influenced verification choices (e.g., one-hot vs. binary FSM encoding for formal tractability) can be evaluated.\n\n---\n\n## Power and Energy Considerations\n\n**INSUFFICIENT EVIDENCE** — Per-core power and energy are not characterized for XH-1.\n\n**ASSUMPTION** — Verification-influenced power decisions (e.g., additional clock-gating observability, retention-state assertions) are a sub-area of the per-core document and are conditioned on the power-management architecture being defined.\n\n**OPEN QUESTION** — Does XH-1 implement RISC-V power-management extensions (e.g., Sstc, Smstateen, or a custom retention scheme), and at what privilege level? Until this is answered, power-aware verification cannot be scoped.\n\n**OPEN QUESTION** — How is energy-aware DV (e.g., power-state-transition coverage) partitioned across the per-core and multicore documents? An explicit ownership line is required.\n\n---\n\n## Implementation Considerations\n\n**OPEN QUESTION** — XH-1 implements a defined subset of the RISC-V ISA (base integer I, and a selected set of M, A, C, F, D, V, B, H, and cryptography sub-extensions under the Zk/Zkn/Zks/Zb* namespace, plus platform extensions). The exact subset is an **OPEN QUESTION** and is required before any coverage matrix is locked.\n\n**OPEN QUESTION** — Is the core in-order or out-of-order? This drives the verification surface:\n\n- In-order: simpler scoreboard, no speculation-recovery paths.\n- Out-of-order: ROB/scoreboard, load-store queue, register-renaming, speculation-recovery, and memory-ordering fences are all first-class verification targets. A formal model of the ROB is typically bounded (small replay queue depth, fixed rename-map width) and tractable within that bound; a full out-of-order memory subsystem is generally beyond unbounded proof.\n\n**OPEN QUESTION** — The ROB formal bound (proof depth, replay-queue depth, rename-map width) is project-owned and is not fixed here. The bound is selected to make the model checker tractable while exercising the recovery paths the project considers in-scope for formal closure. The bounded proof covers the in-bound states; the unbounded remainder is covered by simulation. This division of labor is stated as a PROPOSAL, conditional on the in-order/out-of-order answer and on the project's risk posture.\n\n**ASSUMPTION** — A reference model (Spike-class or SAIL-class) is available for differential checking, and the riscv-arch-test signing flow is adopted so that compliance is portable. Both are working assumptions until confirmed.\n\n**PROPOSAL** — Where the project's extensions include V (vector), B (bitmanip), H (hypervisor), or cryptography sub-extensions (Zk/Zkn/Zks/Zb*), the per-core coverage matrix is extended with extension-specific cross-coverage, and formal proofs on the corresponding functional units are added to the formal scope. The conditional language here is the binding form: the coverage and formal scopes are conditioned on the confirmed ISA profile.\n\n**PROPOSAL** — The riscv-dv generator is integrated as the primary random program source, with hand-written directed sequences retained for microarchitectural corner cases (cache aliasing, TLB refill, fence ordering, atomic RMW, debug-entry/exit, and CSR WARL). [Source: github.com/chipsalliance/riscv-dv.]\n\n---\n\n## Verification Considerations\n\n**PROPOSAL** — A coverage closure plan is defined before any RTL is signed off. Exit criteria are stated in terms of:\n\n- **Functional coverage.** Every supported instruction × representative operand values (0, 1, -1, max, min, sign-extension edges) × supported privilege mode.\n- **Exception/interrupt coverage.** Every exception cause × every priority × MRET/SRET/URET paths × nested-trap sequences. [Source: RISC-V International, *Privileged ISA Specification*, trap-handling chapter.]\n- **CSR coverage.** Every implemented CSR × every writable bit × reset value × WARL semantics. Formal proof of WARL invariants is a sub-goal.\n- **Memory access coverage.** Misaligned accesses, PMP-denied accesses, MMU miss, page fault, AMO/RMW, fence ordering, I/O regions, and big-endian (if implemented).\n- **Reset and debug.** Every reset reason (power-on, warm, retention, debug) and every debug state transition in the RISC-V Debug Specification version confirmed for XH-1.\n- **Formal.** Targeted bounded proofs on bounded sub-modules: CSR file WARL invariants, TLB invariants, exception priority encoder, retire/ROB, debug-transport FSM, trace-encoder FSM. The bound for each sub-module is project-owned (see Implementation Considerations).\n\n**OPEN QUESTION** — The numeric coverage thresholds (legal-bin closure percentage, toggle coverage, assertion coverage, mutation score) are project-owned and are not fabricated here. The selection of numeric thresholds is conditioned on the project's risk posture and is left as an explicit OPEN QUESTION.\n\n**OPEN QUESTION** — Mutation / fault-injection (alternative C3) is a complementary measure of suite strength. Whether it is in or out of the XH-1 sign-off artifact is not decided in this document. Options: (i) include mutation score as an explicit exit criterion; (ii) use mutation as a development-time diagnostic without a numeric exit criterion; (iii) exclude mutation entirely. The decision is deferred to the verification strategy document.\n\n**PROPOSAL** — Extension-specific items in the coverage list are conditioned on the confirmed ISA profile:\n\n- F/D coverage (including NaN-boxing) is included only if the F/D extension is implemented.\n- V (vector) coverage is included only if V is implemented, and is scoped to the implemented VLEN/ELEN and to the chosen vector subset.\n- B (bitmanip) coverage is included only if B is implemented.\n- H (hypervisor) coverage is included only if H is implemented and is paired with a separate virtual-memory coverage plan.\n- Cryptography coverage (Zk, Zkn, Zks, Zbkb, Zbkc, Zbkx) is included only for the implemented sub-extensions, with side-channel considerations deferred to a separate document.\n\nThis conditional scoping is the binding form: the coverage list is fixed at the granularity of *conditioned on the confirmed ISA profile*.\n\n**PROPOSAL** — Coverage merging across flows (simulation, formal, emulation) is required: the formal-tool coverage reports and the simulator coverage reports must be merged into a single per-core dashboard so that the sign-off artifact is unambiguous. At 128-instance scale, the merge is performed per-instance and then aggregated.\n\n**OPEN QUESTION** — Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level? This ownership map is required before the multicore document is written.\n\n---\n\n## Software Considerations\n\n**PROPOSAL** — The per-core verification artifacts are aligned with the downstream software flow:\n\n- The riscv-arch-test signature is producible from the verified core and is delivered alongside the reference manual. At 128-instance scale, the signature-generation step is repeated per instance (or per representative instance, per the project-owned decision; see 128-Core Scalability).\n- A linker script, boot ROM contract, and SBI interface contract are specified in parallel with the core's CSRs, to avoid post-silicon software surprises. [Source: RISC-V International, *SBI Specification*.]\n- A pre-silicon architectural reference (Spike-compatible or SAIL) is published for software teams so that OS porting and toolchain work proceed in parallel with RTL. The specific version of Spike or SAIL is pinned to a commit or release tag as part of the verification strategy document; this document does not commit to a version.\n\n**FACT** — The SBI interface is specified by RISC-V International; pinning to a specific SBI version is required and is an **OPEN QUESTION** for XH-1.\n\n---\n\n## Recommendation\n\n**RECOMMENDATION** — Adopt the hybrid simulation + formal methodology (A2) for per-core verification, with the following binding choices:\n\n1. **Simulation workhorse.** Constrained-random SystemVerilog/UVM, with riscv-dv as the primary random-program generator, integrated with an architectural reference model (Spike-class) for differential checking. The specific simulator and reference-model version are project-owned and are deferred to the verification strategy document (see Open Questions).\n2. **Formal scope.** Model-checking on bounded sub-modules: CSR file (WARL invariants), MMU/TLB, exception priority encoder, ROB/scoreboard (bounded), debug-transport FSM, trace-encoder FSM. Candidate tool categories are commercial model checkers (e.g., Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal) and, for small bounded sub-modules, open-source model checkers (e.g., Yosys + SymbiYosys). Specific tool selection, license procurement, and per-sub-module tool assignment are deferred to the verification strategy document and are not bound by this document.\n3. **Compliance.** riscv-arch-test signature generation is part of the sign-off artifact. At 128-instance scale, the per-instance signature-generation policy is an **OPEN QUESTION** (see 128-Core Scalability).\n4. **Coverage model.** Functional matrix indexed by ISA extension × privilege level × interrupt/exception path, with extension-specific items conditioned on the confirmed ISA profile. Mutation / fault-injection status is deferred to the verification strategy document (see Verification Considerations).\n5. **Collateral reuse.** Per-core verification collateral is constructed as a single instance-independent artifact reused across all 128 instances. Cross-instance scenarios are owned by the multicore verification document.\n6. **Ownership.** Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are owned by the multicore verification document.\n\nThe numeric coverage thresholds, the choice of simulator and formal tool, the specific reference-model version, and the coverage-closure horizon are **OPEN QUESTIONS** and are deferred to the verification strategy document and the project's sign-off policy. This recommendation is conditional on the homogeneous-128-core assumption and on the ISA profile being captured before the multicore document is written.\n\n---\n\n## Confidence\n\nMedium. The methodology recommendation follows from established industry practice and the cited RISC-V specifications; the recommendation does not depend on the unresolved numeric thresholds. Confidence would be upgraded to High once the **OPEN QUESTIONS** below are closed.\n\n---\n\n## Open Questions\n\n- What RISC-V base and extensions does XH-1 implement (RV32/RV64, I/M/A/F/D/C/V/B/H, cryptography sub-extensions under Zk/Zkn/Zks/Zb*, supported privilege modes, hypervisor)?\n- Is the core in-order or out-of-order?\n- Is the 128-core array homogeneous, or are there variants (application cores plus a management/IoT core)?\n- What is the XH-1 target process node, per-core area budget, clock frequency, and tape-out schedule?\n- What simulator and formal-tool stack is available or mandated (commercial, Verilator, cocotb, Yosys + SymbiYosys, custom)?\n- What is the project's stance on the RISC-V International compliance signing process, and at what SBI version is the platform pinned?\n- What numeric coverage exit criteria (legal-bin percentage, toggle, assertion, mutation score) does the project accept for tape-out, and is mutation in or out of the sign-off artifact?\n- Which RISC-V Debug Specification version (0.13 or 1.0.0) and which Trace Specification version (2.0.0 or other) apply to XH-1?\n- Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level?\n- Does XH-1 implement RISC-V power-management extensions, and how is energy-aware DV partitioned across the per-core and multicore documents?\n- Is FPGA prototyping or emulation capacity available at the scale required to boot 128 cores?\n- What is the per-instance riscv-arch-test signature policy (one per instance, one per representative instance, parameterized)?\n- What are the formal proof bounds (depth, state-space width) for the ROB and other bounded sub-modules?\n\n---\n\n## Sources\n\n- RISC-V International, *Unprivileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/\n- RISC-V International, *Privileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/\n- RISC-V International, *Debug Specification*, version 1.0.0. https://riscv.org/technical/specifications/\n- RISC-V International, *Trace Specification*, version 2.0.0. https://riscv.org/technical/specifications/\n- RISC-V International, *Cryptography Extensions* (Zk, Zkn, Zks, Zbkb, Zbkc, Zbkx). https://riscv.org/technical/specifications/\n- RISC-V International, *SBI Specification*. https://riscv.org/technical/specifications/\n- RISC-V International, *Compliance Policy* and *Architecture Test* framework. https://github.com/riscv-arch-test\n- OpenHW Group, *CV32E40P Core Documentation*. https://github.com/openhwgroup/core-cv32e40p\n- OpenHW Group, *CV32E40PX Core Documentation*. https://github.com/openhwgroup/core-cv32e40px\n- Google / CHIPS Alliance, *riscv-dv* (ISA-aware random program generator). https://github.com/chipsalliance/riscv-dv\n- Berkeley / SiFive, *Spike* (riscv-isa-sim). https://github.com/riscv/riscv-isa-sim (no specific version pinned in this document; pinning is deferred to the verification strategy document.)\n- Cambridge, *SAIL RISC-V* (architectural reference model). https://github.com/rems-project/sail-riscv\n- Imperas, *riscvOVPsim* (commercial reference model). https://github.com/riscv-ovpsim\n- Siemens EDA, *JasperGold* formal verification platform (product documentation). https://eda.sw.siemens.com/en-US/ic/jaspergold/\n- Synopsys, *VC Formal* (product documentation). https://www.synopsys.com/verification/static-and-formal-verification/vc-formal.html\n- Synopsys, *360 DV-Formal* (formerly OneSpin 360 DV; OneSpin acquired by Synopsys in 2022). https://www.synopsys.com/verification/static-and-formal-verification/360-dv-formal.html\n- YosysHQ, *Yosys* (open-source synthesis and formal-verification framework). https://github.com/YosysHQ/yosys\n- YosysHQ, *SymbiYosys (SBY)* (open-source formal verification front-end). https://github.com/YosysHQ/sby\n\n`INSUFFICIENT EVIDENCE` is recorded above in the Performance, Area, Power and Energy, and Comparison sections pending the project inputs that close the corresponding **OPEN QUESTIONS**. No numeric claim in any of those sections has been fabricated.","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":10018,"completion_tokens":7761,"total_tokens":17779,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":135,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round4.md b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round4.md new file mode 100644 index 0000000..106de42 --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round4.md @@ -0,0 +1,402 @@ +# Core Verification + +## Document Conventions + +This document uses a fixed label taxonomy applied to every claim: + +- **FACT** — A statement supported by a cited primary or authoritative source. +- **ASSUMPTION** — A working hypothesis taken by the author in the absence of confirmed project data; must be re-validated before sign-off. +- **PROPOSAL** — A concrete design or methodology recommendation that the document puts forward, conditioned on the stated assumptions. +- **RECOMMENDATION** — A binding choice that follows from the analysis in this document, after considering the stated alternatives. +- **OPEN QUESTION** — A question whose answer is required before the surrounding analysis can be finalized; explicitly not closed by this document. + +Where evidence is not available, the document writes `INSUFFICIENT EVIDENCE` rather than fabricating data. + +--- + +## Status + +Substantive research draft. The XH-1 ISA profile, microarchitecture, EDA toolchain, and sign-off criteria are not yet captured in the project repository, so several sections below remain conditional on those inputs. Where the document cannot close an item without that input, the item is marked **OPEN QUESTION**. + +--- + +## Abstract + +XH-1 is a custom 128-core RISC-V design. The verification burden scales sub-linearly with core count only if the per-core verification collateral is instance-independent (a single testbench and coverage model reused for every replicated instance) and the cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are explicitly delegated to a separate multicore verification document. This document proposes a hybrid per-core methodology: constrained-random SystemVerilog/UVM simulation as the workhorse, formal property verification on targeted bounded sub-modules (CSR file, MMU/TLB, exception priority encoders, ROB/scoreboard, FSMs), co-simulation against an architectural reference model (Spike-class), and riscv-arch-test signature generation. Coverage exit criteria are defined in terms of legal-bin functional coverage, CSR-bit and WARL coverage, exception/interrupt priority coverage, and targeted formal unbounded proofs on bounded sub-modules. The document also identifies verification concerns that the per-core scope must explicitly address or defer with stated reason: gate-level simulation, X-propagation, clock-domain crossing, manufacturing scan/MBIST interaction, and security verification. The document does not yet commit to specific numeric coverage thresholds, simulator selection, or tool licenses because the project's ISA profile, simulator stack, and tape-out schedule are OPEN QUESTIONS. + +--- + +## Research Question + +The specific research questions addressed are: + +1. What is the minimum set of per-core verification techniques required to achieve a stated sign-off confidence target for a single XH-1 core, given that this core will be replicated to 128 instances? +2. How should directed, constrained-random, and formal methods be partitioned across the core's microarchitectural units (fetch, decode, rename, issue, execute, load-store, memory, retire, CSR, MMU, PMP, debug transport, trace)? +3. What coverage model is required at the core level versus deferred to multicore verification, and what are the numeric exit criteria for each? +4. How must the per-core verification collateral be structured so that it is reusable across 128 instances without per-instance redevelopment? +5. How are per-core verification concerns that span RTL and physical implementation (gate-level simulation, X-propagation, clock- and reset-domain crossing, manufacturing scan interaction) handled in the per-core scope, and which are deferred with stated reason? + +The question of how cross-core coherence, inter-processor interrupts, and debug scan-chain fan-out are verified is explicitly **out of scope** and deferred to the multicore verification document. + +--- + +## Background + +**FACT** — The RISC-V ISA is defined by two principal specifications maintained by RISC-V International: the *Unprivileged ISA Specification* and the *Privileged ISA Specification*. Both are ratified, versioned, and publicly available. [Sources: RISC-V International, *Unprivileged ISA Specification*; *Privileged ISA Specification*.] + +**FACT** — The RISC-V *Debug Specification* is at version 1.0.0 (ratified). [Source: RISC-V International, *RISC-V Debug Specification*, version 1.0.0, https://riscv.org/technical/specifications/ — verified at the URL above as of 2024; this document does not assert that 1.0.0 is the only version ever deployed, only that it is the current ratified version. The Debug 0.13 specification existed historically; whether 0.13 is observed in deployed cores is not stated as FACT in this document. See Open Questions for XH-1's specific Debug version.] + +**FACT** — The RISC-V *Trace Specification* is at version 2.0.0 as of the most recent public release prior to this document. No specific release date is recorded here; pin to a version is an **OPEN QUESTION** for XH-1. [Source: RISC-V International, *RISC-V Trace Specification*, version 2.0.0, https://riscv.org/technical/specifications/.] + +**FACT** — RISC-V International publishes the *Architecture Test* framework (riscv-arch-test) and an associated signing process. Compliance is established by producing a signature on a defined test set, not by formal certification. [Source: github.com/riscv-arch-test, RISC-V International *Compliance Policy*.] + +**FACT** — OpenHW Group's CV32E40P is a 32-bit, in-order, RV32IMFC RISC-V core. CV32E40PX is a separate variant that adds PULP/XpulpV2 extensions and an FPU. The two are not interchangeable. [Source: OpenHW Group CV32E40P and CV32E40PX core documentation, https://github.com/openhwgroup/core-cv32e40p and https://github.com/openhwgroup/core-cv32e40px.] (Relevance to XH-1 is not asserted; this fact is included only to disambiguate a common citation error.) + +**FACT** — The RISC-V extension taxonomy as ratified by RISC-V International separates unprivileged and privileged extensions: + +- The unprivileged ISA general-purpose extensions are designated by single capital letters: I (base integer), M (integer multiply/divide), A (atomic), F (single-precision floating-point), D (double-precision floating-point), C (compressed). [Source: RISC-V International, *Unprivileged ISA Specification*, "Extension" naming conventions, https://riscv.org/technical/specifications/.] +- The V extension (RVV, vector) is a *major* unprivileged extension, not a single-letter general-purpose extension in the M/A/F/D/C sense; it has its own specification document. [Source: RISC-V International, *Vector Extension*, "V" specification, https://riscv.org/technical/specifications/.] +- B (bitmanip) is not a single extension in the current taxonomy. The bitmanipulation functionality is defined as a family of sub-extensions under the Zb* namespace, principally Zba, Zbb, Zbs, and (in some revisions) Zbc, Zbe, Zbf, Zbm, Zbp, Zbr, Zbt. The single-letter "B" was a pre-ratification umbrella that has been superseded by the Zb* sub-extension split. [Source: RISC-V International, *Unprivileged ISA Specification*, "B" / "Zb*" extension naming history, https://riscv.org/technical/specifications/.] +- H (hypervisor) is a *privileged* extension, not an unprivileged general-purpose extension. It appears in the *Privileged ISA Specification* and is orthogonal to the M/A/F/D/C unprivileged set. [Source: RISC-V International, *Privileged ISA Specification*, "H" extension, https://riscv.org/technical/specifications/.] +- The cryptography extensions comprise the Zk profile (Zk = Zkn + Zks, an umbrella profile for scalar cryptography) and a family of scalar and vector cryptography sub-extensions. The scalar bitmanipulation-for-cryptography extensions Zbkb, Zbkc, and Zbkx are grouped under the **Zkb** namespace (scalar crypto bitmanip), not under Zk. Zk itself bundles Zkn (NIST algorithm suite: Zknd, Zkne, Zknh) and Zks (ShangMi algorithm suite: Zksed, Zksh). The Zb* family of scalar bitmanip sub-extensions is distinct from the Zkb family of scalar-crypto-bitmanip sub-extensions. [Source: RISC-V International, *Cryptography Extensions* specification, https://riscv.org/technical/specifications/.] + +**ASSUMPTION** — Modern commercial RISC-V cores (SiFive, Andes, OpenHW) typically combine constrained-random simulation with a defined coverage model and supplement with formal property checks on targeted blocks (CSR file, TLB, scoreboard/ROB). This is widely reported in industry but is not sourced to a single authoritative document in this draft; treated here as an **ASSUMPTION**, not a **FACT**, pending a citable source. + +**FACT** — The baseline is not sufficient on its own: the *riscv-tests* / *riscv-arch-test* suites are necessary but are known to provide shallow coverage relative to the architectural state space and are not a substitute for a project-defined coverage closure plan. [Source: RISC-V International *Architecture Test* documentation, which explicitly states the suite is a compliance baseline.] + +--- + +## Existing Approaches + +**FACT** — The following categories of core-level verification flow are established in the EDA and RISC-V ecosystems: + +1. Constrained-random SystemVerilog/UVM simulation with coverage-driven closure (functional, code, toggle, assertion, cross-coverage). +2. Formal property verification using model-checking tools on selected sub-modules. Correctly categorized tools in this space include: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal (formerly OneSpin 360 DV; OneSpin was acquired by Synopsys in 2022 and the product is now positioned as part of the Synopsys formal portfolio), and open-source flows such as Yosys + SymbiYosys (SBY). [Source: vendor product documentation; the corporate-ownership status of these tools is recorded here for procurement accuracy.] +3. FPGA prototyping and emulation acceleration (e.g., Synopsys ZeBu, Cadence Palladium, in-house FPGA boards). +4. Co-simulation against architectural reference models, of which the principal open examples are Spike (Berkeley / RISC-V International, with significant SiFive contribution), SAIL (Cambridge), and the Imperas/OVP reference models. [Sources: github.com/riscv/riscv-isa-sim (Spike); github.com/rems-project/sail-riscv; github.com/riscv-ovpsim.] +5. Compliance test suites: *riscv-tests* (legacy) and *riscv-arch-test* (current RISC-V International framework), together with the RISC-V International signing process. [Source: github.com/riscv-arch-test.] +6. Mutation-based verification tools (e.g., Certitude, proprietary in-house mutation frameworks). +7. Coverage merging across simulators and prototypes (e.g., Synopsys VCS + ZeBu, Cadence Xcelium + Palladium). +8. Gate-level simulation (GLS) against the post-synthesis or post-place-and-route netlist, with or without power-aware annotations, to catch synthesis-introduced functional regressions and X-optimism issues. [Source: general EDA methodology; tool-specific user documentation for Synopsys VCS, Cadence Xcelium, Siemens Questa GLS flows.] +9. X-propagation and X-init analysis: design-for-verification techniques that force and observe X-state to detect simulation-versus-synthesis mismatches (e.g., Synopsys X-Prop, Cadence X-Cell, Siemens Questa X-Prop). [Source: vendor methodology documentation.] +10. Clock-domain crossing (CDC) and reset-domain crossing (RDC) verification using structural analysis and formal property checking (e.g., Synopsys SpyGlass CDC / RDC, Cadence JasperGold CDC, Siemens Questa CDC). [Source: vendor methodology documentation.] + +--- + +## Alternative Designs + +The principal methodology alternatives considered are: + +- **A1. Pure simulation-driven sign-off.** Constrained-random UVM with full coverage closure; no formal. Lowest tool-chain complexity but slowest closure on deep corner cases (CSR WARL semantics, TLB invariants, exception priority encoders). +- **A2. Hybrid simulation + formal (selected).** Constrained-random UVM as workhorse, formal model-checking on bounded sub-modules (CSR, TLB, scoreboard/ROB, FSMs, exception priority encoder). Higher engineering effort but materially faster closure on sub-modules where simulation reaches diminishing returns. +- **A3. Formal-first, simulation-supplemental.** Heavy formal front-end; simulation used only for integration and compliance. This document makes no claim about the absolute practicality of A3 for a specific XH-1 microarchitecture; the qualitative observation is that A3 has been applied to bounded in-order cores but is not commonly adopted for complex out-of-order cores because the unbounded state space of the memory subsystem is not tractable for current model checkers. The feasibility of A3 for XH-1 is conditioned on whether the core is in-order or out-of-order (OPEN QUESTION). **ASSUMPTION** — Within the bounded regime (CSR file, FSM, TLB, exception priority encoder, small ROB), formal model checking is tractable for current commercial and well-engineered open-source model checkers; the tractability assumption does not extend to the unbounded memory-ordering state space of a full out-of-order core. +- **A4. Emulation/FPGA-first.** Boots real software (Linux, distro, stress workloads) early. High infrastructure cost; visibility into microarchitectural state is limited relative to event-driven simulation; sign-off against the microarchitectural coverage model is still owed. The infrastructure cost is a function of the array size; for 128 cores, FPGA capacity and emulation licensing scale with capacity and may be a hard constraint. Capacity availability for XH-1 is an **OPEN QUESTION**. + +Within the simulation domain: + +- **B1. In-house UVM testbench.** Full control, no license cost, but high engineering effort. +- **B2. Commercial verification IP.** Faster start; recurring license cost; opaque to RTL changes. +- **B3. Generator-based flows (e.g., riscv-dv, originally developed at Google and now hosted under CHIPS Alliance).** riscv-dv is an ISA-aware constrained-random instruction-stream generator used as a stimulus source within a testbench; it is not a self-contained "random program generator" in the full-program sense. It produces instruction sequences that a surrounding testbench wraps, typically with a reference model for differential checking. [Source: github.com/chipsalliance/riscv-dv.] + +Within the formal-tool domain: + +- **F1. Commercial model checkers.** Examples: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal. Higher capacity, commercial support, license cost. +- **F2. Open-source model checkers.** Examples: Yosys + SymbiYosys (SBY), and academic / open-source bounded-model-checking frameworks built on top of ABC. Lower cost; smaller capacity; smaller ecosystem. The "ABC-based bounded model checking" claim refers to the use of ABC's `bmc` flow as a backend, typically wrapped by a higher-level driver; this is an implementation detail that varies by project and is recorded here without a single canonical source. [Source: https://github.com/YosysHQ/yosys; https://github.com/YosysHQ/sby; ABC project documentation, https://github.com/yosyshq/abc.] + +Coverage-model alternatives: + +- **C1. Structural only (line/branch/toggle).** Fast to collect; known to miss functional holes. +- **C2. Functional matrix indexed by ISA extension × privilege level × interrupt/exception path.** Necessary for sign-off; higher collection and debug cost. +- **C3. Functional + mutation score.** Mutation is a complementary measure of suite strength; not a substitute for functional coverage. Whether mutation is part of the XH-1 sign-off artifact is an **OPEN QUESTION** (see Verification Considerations). + +--- + +## Comparison + +**INSUFFICIENT EVIDENCE** for a quantitative comparison. The relative run time, license cost, engineering effort, coverage-closure time, and debug productivity of A1–A4, B1–B3, and F1–F2 depend on the chosen EDA stack, the ISA profile, and the in-house team's existing code base, none of which are captured in the project repository. A numeric trade-off table is not fabricated here. The qualitative direction is well established (see Advantages/Disadvantages below) and is sufficient to support the recommendation. + +--- + +## Advantages + +**Hybrid simulation + formal (A2):** + +- Formal closes deep corner cases (CSR WARL, TLB invariants, exception priority encoder) faster than simulation, where simulation reaches diminishing returns. (Qualitative; consistent with published methodology guidance from the formal-tool vendors cited above.) +- Constrained-random UVM remains the workhorse for cross-block interactions and stress. +- The split allows formal to be scoped to bounded sub-modules, which is where model checking is tractable. + +**Generator-based flow (B3, e.g., riscv-dv):** + +- Open-source and ISA-aware; reduces hand-written sequence count. +- Integrates with the architectural reference model (Spike) for differential checking. [Source: github.com/chipsalliance/riscv-dv; github.com/riscv/riscv-isa-sim.] + +**Open-source formal (F2):** + +- No license cost; can be applied to small bounded sub-modules (single FSMs, small CSR files) without commercial procurement. +- Useful as a pre-screen or as a fallback for sub-modules where commercial formal capacity is not warranted. + +**Compliance framework:** + +- riscv-arch-test with the RISC-V International signing process produces a portable, vendor-neutral artifact that downstream software/OS teams can use. [Source: github.com/riscv-arch-test; RISC-V International *Compliance Policy*.] + +--- + +## Disadvantages + +**Hybrid simulation + formal (A2):** + +- Two tool flows must be integrated, which adds engineering effort and requires a coverage-merging strategy across flows. +- Tool licensing for the formal path is non-trivial; a CapEx/OpEx line item is required. + +**Generator-based flow (B3):** + +- riscv-dv is RV32/RV64 ISA-focused; non-ISA RTL (custom accelerators, custom CSRs) is not generated and must be hand-written. + +**Compliance framework:** + +- The riscv-arch-test suite is shallow; producing a signature on it is necessary but not sufficient for sign-off. Treating it as sufficient is a known failure mode. (This is a usage guideline, not a property of the framework itself.) + +**Open-source formal (F2):** + +- Lower model-checking capacity than commercial tools; not all sub-modules will close. +- Smaller ecosystem of commercial support. + +**General across alternatives:** + +- None of the alternatives in isolation produces a tape-out-ready sign-off artifact; all require an explicit, project-owned coverage-closure plan with numeric exit criteria. + +--- + +## XH-1 Considerations + +**ASSUMPTION** — XH-1 is a custom 128-core RISC-V design. Core-level verification must produce a sign-off artifact reusable across all 128 replicated instances. + +**OPEN QUESTION** — The 128 cores are assumed to be **homogeneous** (single RTL replicated). Whether this holds, or whether XH-1 includes variants (e.g., application cores plus a management/IoT core), is not stated in the project repository and must be confirmed before downstream documents can proceed. + +**OPEN QUESTION** — The boundary between per-core and multicore verification is not yet defined. Candidates: + +- Per-core: L1 I-cache, L1 D-cache, MMU/TLB, PMP, CSR, debug transport module, trace encoder, retire/ROB, exception/interrupt controller. +- Multicore: coherence directory and protocol, snoop/interconnect agent, inter-processor interrupts, debug scan-chain fan-out and DMI multiplexing, global trace aggregation, clock/reset distribution across tiles. + +**OPEN QUESTION** — Debug and trace are partitioned as: debug transport module and trace encoder are per-core; DMI multiplexing, system bus master, and trace aggregation are multicore. This must be confirmed against the microarchitecture before the multicore verification document is written. + +**OPEN QUESTION** — The RISC-V Debug Specification version that applies to XH-1 is not stated. The current ratified version is 1.0.0; the project's choice between 1.0.0 and any earlier revision (including 0.13, which was a pre-ratification draft) is not captured. Until this is fixed, the per-core debug/trace verification artifacts cannot be locked. The Trace Specification version is similarly unresolved (current ratified version is 2.0.0). + +--- + +## 128-Core Scalability + +**PROPOSAL** — Per-core verification collateral (testbench, sequences, assertions, coverage models) is constructed as a **single instance-independent artifact** that is run for each of the 128 replicated cores (or, equivalently, run once with a parameterizable instance handle). **ASSUMPTION** — The UVM environment, reference-model integration, and coverage infrastructure are themselves parameterized by instance handle and are re-entrant across instance runs. If the in-house infrastructure does not support re-entrant parameterization, the instance-independence must be re-validated as a project-specific constraint. Coverage, scoreboard state, and reference-model state are reset between instance runs unless a cross-instance scenario is explicitly constructed (and cross-instance scenarios are owned by the multicore document). This is the same artifact exercised 128 times, not 128 specialized artifacts. + +**PROPOSAL** — The formal-tool bound (proof depth, state-space depth, number of register bits) is selected such that the same bound applies to all 128 instances; the bound is not a function of the instance index. The formal collateral is therefore also instance-independent. + +**ASSUMPTION** — The 128-core design reuses a single verified core RTL. If the array is heterogeneous, the instance-independence collapses to one per variant, and the per-core effort is multiplied by the number of variants. + +**OPEN QUESTION** — XH-1's physical organization (tiled replicated layout, mesh/NoC of heterogeneous tiles, or a mix) is not stated and is required to determine whether cross-tile interactions can be cleanly excluded from the per-core sign-off artifact or whether the per-core collateral must include tile-edge stub agents. + +**PROPOSAL** — Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out) are **out of scope** for the per-core document and are owned by the multicore verification document. (This is a PROPOSAL based on the verification-reasoning argument that mixing per-core and multicore concerns produces a fragile, instance-coupled artifact; it is not a "known anti-pattern" in the sense of an externally codified rule.) + +**OPEN QUESTION** — How is the riscv-arch-test signing flow extended to 128 instances? The compliance flow is per-core; the project intent (one signature per instance, one signature from a representative instance, or a parameterized signature) is not stated. + +**PROPOSAL** — Regression parallelism across 128 instances is achieved by distributing identical per-core runs across the host farm. **ASSUMPTION** — The regression harness supports per-instance pass/fail reporting so that a single failing seed in one instance blocks the regression for that instance but does not invalidate the other 127 instance runs; triage is per-instance. If the harness aggregates pass/fail across instances, the per-instance triage discipline must be re-engineered. Cross-instance coverage merging is performed by aggregating the per-instance coverage databases into a single per-core dashboard. + +**OPEN QUESTION** — FPGA prototyping or emulation capacity for 128 cores is not stated. A4 (Emulation/FPGA-first) is feasible only if the platform can host the full array; the alternative is to emulate a single instance plus a representative tile-edge stub, which is a per-core artifact, not a multicore one. + +--- + +## Performance Considerations + +**OPEN QUESTION** — Core-level simulation throughput (instructions/second, coverage points/hour) depends on the chosen simulator (commercial event-driven, Verilator, cocotb, or custom), the workload, and the host farm, none of which are fixed. No numeric simulation throughput is fabricated here. + +**ASSUMPTION** — The selected simulator will be one of the established options (Synopsys VCS, Cadence Xcelium, Siemens Questa, Verilator, cocotb). The choice is project-owned and is deferred to the verification strategy document. + +**OPEN QUESTION** — The trade-off between open-source simulators (Verilator, cocotb) and commercial event-driven simulators at 128-core regression scale is not characterized here. The relevant axes are: per-instance throughput, license cost, UVM support, coverage-tool integration, and debug visibility. A quantitative trade-off depends on the chosen farm size and is not fabricated. + +**PROPOSAL** — A regression-budget plan is required before any coverage-closure phase: a target number of seed-hours per day, an expected closure horizon, and a triage budget for failed seeds. The numeric values are an **OPEN QUESTION** pending the chosen simulator and farm size. + +--- + +## Area Considerations + +**INSUFFICIENT EVIDENCE** — Verification-influenced area decisions (adding observability or assertion logic, formal-friendly state encodings, scan/DFT hooks) are not characterized for XH-1. + +**ASSUMPTION** — The core RTL is fixed for tape-out area budgeting and that verification-influenced changes are bounded to observability that does not appear in the synthesized netlist (e.g., `ifdef` guards). This must be confirmed against the synthesis/scan insertion flow. + +**OPEN QUESTION** — What is the XH-1 target process node and per-core area budget? Both are required before area-influenced verification choices (e.g., one-hot vs. binary FSM encoding for formal tractability) can be evaluated. + +--- + +## Power and Energy Considerations + +**INSUFFICIENT EVIDENCE** — Per-core power and energy are not characterized for XH-1. + +**ASSUMPTION** — Verification-influenced power decisions (e.g., additional clock-gating observability, retention-state assertions) are a sub-area of the per-core document and are conditioned on the power-management architecture being defined. + +**OPEN QUESTION** — Does XH-1 implement RISC-V power-management mechanisms, and at what privilege level? The RISC-V mechanisms relevant to this question include: + +- WFI (`wait-for-interrupt`) and the WFI-idle semantics defined in the *Privileged ISA Specification*; these are the standardized low-power entry points. [Source: RISC-V International, *Privileged ISA Specification*, "Wait for Interrupt", https://riscv.org/technical/specifications/.] +- `Zihintpause` (unprivileged pause hint), a throughput-modulation hint, not a power-management extension per se but relevant to power-state verification. [Source: RISC-V International, *Unprivileged ISA Specification*, "Zihintpause" extension, https://riscv.org/technical/specifications/.] +- Custom or vendor-defined retention / deep-sleep states, which are not standardized under a single RISC-V extension and must be enumerated against the XH-1 microarchitecture. + +Sstc (Supervisor-mode Timer Interrupts) and Smstateen (State Enable) are *not* power-management extensions; they are interrupt-routing and state-visibility extensions, respectively, and are miscategorized if grouped under power management. [Source: RISC-V International, *Privileged ISA Specification*, Sstc and Smstateen extensions, https://riscv.org/technical/specifications/.] Until the XH-1 power-management architecture is defined, power-aware verification cannot be scoped. + +**OPEN QUESTION** — How is energy-aware DV (e.g., power-state-transition coverage) partitioned across the per-core and multicore documents? An explicit ownership line is required. + +--- + +## Implementation Considerations + +**OPEN QUESTION** — XH-1 implements a defined subset of the RISC-V ISA (base integer I, and a selected set of M, A, C, F, D, the major V extension, the Zb* bitmanipulation sub-extensions, the H privileged extension, and cryptography sub-extensions under Zkn, Zks, and Zkb, plus platform extensions). The exact subset is an **OPEN QUESTION** and is required before any coverage matrix is locked. + +**OPEN QUESTION** — Is the core in-order or out-of-order? This drives the verification surface: + +- In-order: simpler scoreboard, no speculation-recovery paths. +- Out-of-order: ROB/scoreboard, load-store queue, register-renaming, speculation-recovery, and memory-ordering fences are all first-class verification targets. **ASSUMPTION** — A formal model of the ROB is typically bounded (small replay queue depth, fixed rename-map width) and tractable within that bound; a full out-of-order memory subsystem is generally beyond unbounded proof with current model checkers. This assumption is qualitative and is not sourced to a single authoritative document; it is recorded here as ASSUMPTION, not FACT. + +**OPEN QUESTION** — The ROB formal bound (proof depth, replay-queue depth, rename-map width) is project-owned and is not fixed here. The bound is selected to make the model checker tractable while exercising the recovery paths the project considers in-scope for formal closure. The bounded proof covers the in-bound states; the unbounded remainder is covered by simulation. This division of labor is stated as a PROPOSAL, conditional on the in-order/out-of-order answer and on the project's risk posture. + +**ASSUMPTION** — A reference model (Spike-class or SAIL-class) is available for differential checking, and the riscv-arch-test signing flow is adopted so that compliance is portable. Both are working assumptions until confirmed. + +**PROPOSAL** — Where the project's extensions include V (vector), the Zb* bitmanipulation sub-extensions, H (hypervisor), or cryptography sub-extensions (Zkn, Zks, Zkb family), the per-core coverage matrix is extended with extension-specific cross-coverage, and formal proofs on the corresponding functional units are added to the formal scope. The conditional language here is the binding form: the coverage and formal scopes are conditioned on the confirmed ISA profile. + +**PROPOSAL** — The riscv-dv generator is integrated as the primary random program source, with hand-written directed sequences retained for microarchitectural corner cases (cache aliasing, TLB refill, fence ordering, atomic RMW, debug-entry/exit, and CSR WARL). [Source: github.com/chipsalliance/riscv-dv.] + +**OPEN QUESTION** — PMP is listed under per-core scope. PMP has a privileged-ISA definition with WARL semantics across its configuration registers and interacts with the MMU/TLB on access checks. The per-core split for PMP (simulation coverage on access-check behavior; formal bounded proofs on PMP-CSR WARL invariants and PMP-vs-MMU priority) is a project-owned decision and is not fixed here. + +--- + +## Cross-Cutting Verification Concerns + +The per-core scope must explicitly address or defer with stated reason the following cross-cutting concerns. Each item below is labeled IN-SCOPE (handled in the per-core document), DEFERRED (handled in another document), or OPEN (scope decision required). + +- **Gate-level simulation (GLS).** **OPEN.** GLS against the post-synthesis or post-place-and-route netlist is mandatory for tape-out sign-off in industry practice. Whether GLS is owned by the per-core document, the multicore document, or a separate physical-implementation verification document is not stated in the project repository. **ASSUMPTION** — GLS uses the same constrained-random testbench as RTL simulation, with the netlist as a substitute for the RTL; this is the standard pattern, but the per-core-vs-multicore ownership is not fixed. + +- **X-propagation and X-init analysis.** **OPEN.** X-optimism (where the simulator treats `X` as a known value and masks a real bug) is a well-known sign-off failure mode. Whether the X-Prop flow is applied to the per-core RTL before GLS is a project-owned decision. **ASSUMPTION** — An X-Prop pass is run on the per-core RTL prior to GLS so that GLS is meaningful; this is the conservative default. + +- **Clock-domain crossing (CDC) and reset-domain crossing (RDC).** **OPEN.** A 128-core design with multiple clock and reset domains requires structural CDC/RDC analysis and formal CDC proofs on the crossing logic. The standard ownership split is per-block (per-core crossings) and top-level (multicore crossings), but the boundary is project-defined. **ASSUMPTION** — CDC/RDC for crossings *within* a single core (e.g., debug vs. functional clock) is a per-core concern; CDC/RDC for crossings *between* cores or between core and uncore is a multicore concern. + +- **Manufacturing scan / MBIST / LBIST.** **OPEN.** Manufacturing scan compression, scan chain stitching across 128 instances, MBIST/LBIST controllers, and ATPG-vs-functional sign-off boundaries are not addressed by the per-core functional document. Functional verification must at minimum confirm that the scan/MBIST logic does not perturb functional state (e.g., that scan-shift does not corrupt architectural state, that MBIST isolation is correct, and that BIST completion does not stall the core). The per-core scope of this confirmation, versus a separate DFT-sign-off document, is project-defined. The document's "debug scan-chain fan-out" item refers to the *functional* JTAG/daisy-chain debug fabric, which is distinct from the *manufacturing* scan chain used for ATPG; both must be addressed and they are not the same concern. + +- **Security verification.** **OPEN.** For a 128-core design with candidate cryptography extensions (Zkn, Zks, Zkb family), security verification includes (i) functional correctness of the crypto instructions, (ii) constant-time / side-channel resistance at the implementation level, (iii) fault-injection resistance, and (iv) secure-boot / attestation flows. Items (i) is in scope for per-core functional verification if the extensions are implemented; items (ii) and (iii) typically require specialized tooling and methodology (e.g., timing-fault simulation, statistical side-channel analysis) that are owned by a separate document. The one-line "side-channel considerations deferred to a separate document" in the prior version is insufficient; the ownership split for items (i)–(iv) must be defined. + +- **FP corner cases (F/D extension).** **OPEN.** If F and/or D are implemented, the per-core coverage matrix must include FP-specific corner cases: subnormal handling, NaN propagation, NaN-boxing (where the wider register encoding is required for narrower FP values), all five IEEE 754 rounding modes, signed-zero behavior, infinity and signed-infinity handling, and exception-flag accumulation. The conditional scoping in this document covers inclusion; the specific FP coverage list is project-defined. + +- **V (vector) extension coverage.** **OPEN.** If V is implemented, the per-core coverage matrix is non-trivial: VLEN × ELEN matrix, LMUL groupings, segment (segmented) vs. strided vs. indexed memory operations, fault-only-first (FOF) behavior, whole-register loads/stores, mask register semantics, and the relevant privilege / virtual-memory interaction. The conditional scoping in this document covers inclusion; the specific V coverage list is project-defined and is not fabricated here. + +--- + +## Verification Considerations + +**PROPOSAL** — A coverage closure plan is defined before any RTL is signed off. Exit criteria are stated in terms of: + +- **Functional coverage.** Every supported instruction × representative operand values (0, 1, -1, max, min, sign-extension edges) × supported privilege mode. +- **Exception/interrupt coverage.** Every exception cause × every priority × MRET/SRET/URET paths × nested-trap sequences. [Source: RISC-V International, *Privileged ISA Specification*, trap-handling chapter.] +- **CSR coverage.** Every implemented CSR × every writable bit × reset value × WARL semantics. Formal proof of WARL invariants is a sub-goal. +- **Memory access coverage.** Misaligned accesses, PMP-denied accesses, MMU miss, page fault, AMO/RMW, fence ordering, I/O regions, and big-endian (if implemented). +- **Reset and debug.** Every reset reason (power-on, warm, retention, debug) and every debug state transition in the RISC-V Debug Specification version confirmed for XH-1. +- **Formal.** Targeted bounded proofs on bounded sub-modules: CSR file WARL invariants, TLB invariants, exception priority encoder, retire/ROB, debug-transport FSM, trace-encoder FSM. The bound for each sub-module is project-owned (see Implementation Considerations). + +**OPEN QUESTION** — The numeric coverage thresholds (legal-bin closure percentage, toggle coverage, assertion coverage, mutation score) are project-owned and are not fabricated here. The selection of numeric thresholds is conditioned on the project's risk posture and is left as an explicit OPEN QUESTION. + +**OPEN QUESTION** — Mutation / fault-injection (alternative C3) is a complementary measure of suite strength. Whether it is in or out of the XH-1 sign-off artifact is not decided in this document. Options: (i) include mutation score as an explicit exit criterion; (ii) use mutation as a development-time diagnostic without a numeric exit criterion; (iii) exclude mutation entirely. The decision is deferred to the verification strategy document. + +**PROPOSAL** — Extension-specific items in the coverage list are conditioned on the confirmed ISA profile: + +- F/D coverage (including NaN-boxing, subnormal handling, rounding modes) is included only if the F/D extension is implemented. +- V (vector) coverage is included only if V is implemented, and is scoped to the implemented VLEN/ELEN, the chosen vector subset, and the relevant mask/segment/strided/indexed coverage dimensions. +- Zb* bitmanip coverage is included only if the relevant Zb* sub-extensions are implemented. +- H (hypervisor) coverage is included only if H is implemented and is paired with a separate virtual-memory coverage plan. +- Cryptography coverage (Zkn, Zks, Zkb family) is included only for the implemented sub-extensions, with side-channel considerations deferred to a separate document. + +This conditional scoping is the binding form: the coverage list is fixed at the granularity of *conditioned on the confirmed ISA profile*. + +**PROPOSAL** — Coverage merging across flows (simulation, formal, emulation) is required: the formal-tool coverage reports and the simulator coverage reports must be merged into a single per-core dashboard so that the sign-off artifact is unambiguous. At 128-instance scale, the merge is performed per-instance and then aggregated. + +**OPEN QUESTION** — Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level? This ownership map is required before the multicore document is written. + +--- + +## Software Considerations + +**PROPOSAL** — The per-core verification artifacts are aligned with the downstream software flow: + +- The riscv-arch-test signature is producible from the verified core and is delivered alongside the reference manual. At 128-instance scale, the signature-generation step is repeated per instance (or per representative instance, per the project-owned decision; see 128-Core Scalability). +- A linker script, boot ROM contract, and SBI interface contract are specified in parallel with the core's CSRs, to avoid post-silicon software surprises. [Source: RISC-V International, *SBI Specification*.] +- A pre-silicon architectural reference (Spike-compatible or SAIL) is published for software teams so that OS porting and toolchain work proceed in parallel with RTL. The specific version of Spike or SAIL is pinned to a commit or release tag as part of the verification strategy document; this document does not commit to a version. + +**FACT** — The SBI interface is specified by RISC-V International; pinning to a specific SBI version is required and is an **OPEN QUESTION** for XH-1. + +--- + +## Recommendation + +The recommendation below states the binding methodological choices and is conditional on the homogeneous-128-core assumption and on the ISA profile being captured before the multicore document is written. Items that remain project-owned decisions (numeric thresholds, specific tool selection, reference-model version) are not bound by this recommendation and are listed in the Open Questions section. + +**RECOMMENDATION** — Adopt the hybrid simulation + formal methodology (A2) for per-core verification, with the following binding choices: + +1. **Simulation workhorse.** Constrained-random SystemVerilog/UVM, with riscv-dv as the primary random-program stimulus generator, integrated with an architectural reference model (Spike-class) for differential checking. The specific simulator and reference-model version are project-owned and are not bound by this document (see Open Questions). +2. **Formal scope.** Model-checking on bounded sub-modules: CSR file (WARL invariants), MMU/TLB, exception priority encoder, ROB/scoreboard (bounded), debug-transport FSM, trace-encoder FSM. Candidate tool categories are commercial model checkers (e.g., Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal) and, for small bounded sub-modules, open-source model checkers (e.g., Yosys + SymbiYosys). Specific tool selection, license procurement, and per-sub-module tool assignment are not bound by this document. +3. **Compliance.** riscv-arch-test signature generation is part of the sign-off artifact. At 128-instance scale, the per-instance signature-generation policy is an **OPEN QUESTION** (see 128-Core Scalability). +4. **Coverage model.** Functional matrix indexed by ISA extension × privilege level × interrupt/exception path, with extension-specific items conditioned on the confirmed ISA profile. +5. **Collateral reuse.** Per-core verification collateral is constructed as a single instance-independent artifact reused across all 128 instances, subject to the ASSUMPTION that the UVM/reference-model/coverage infrastructure supports parameterizable instance handles. Cross-instance scenarios are owned by the multicore verification document. +6. **Ownership.** Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are owned by the multicore verification document. +7. **Cross-cutting concerns.** The per-core document's scope over GLS, X-Prop, CDC/RDC, manufacturing scan/MBIST interaction, and security verification is an **OPEN QUESTION** and is enumerated in the Cross-Cutting Verification Concerns section. These items are not bound by this recommendation and require an explicit ownership decision before tape-out planning. + +The numeric coverage thresholds, the choice of simulator and formal tool, the specific reference-model version, and the coverage-closure horizon are **OPEN QUESTIONS** and are deferred to the verification strategy document and the project's sign-off policy. + +--- + +## Confidence + +Medium. The methodology recommendation follows from established industry practice and the cited RISC-V specifications; the recommendation does not depend on the unresolved numeric thresholds. Confidence would be upgraded to High once the **OPEN QUESTIONS** below are closed. + +--- + +## Open Questions + +- What RISC-V base and extensions does XH-1 implement (RV32/RV64, I/M/A/F/D/C, the major V extension, the Zb* bitmanipulation sub-extensions, the H privileged extension, the cryptography sub-extensions under Zkn/Zks/Zkb, supported privilege modes, hypervisor)? +- Is the core in-order or out-of-order? +- Is the 128-core array homogeneous, or are there variants (application cores plus a management/IoT core)? +- What is the XH-1 target process node, per-core area budget, clock frequency, and tape-out schedule? +- What simulator and formal-tool stack is available or mandated (commercial, Verilator, cocotb, Yosys + SymbiYosys, custom)? +- What is the project's stance on the RISC-V International compliance signing process, and at what SBI version is the platform pinned? +- What numeric coverage exit criteria (legal-bin percentage, toggle, assertion, mutation score) does the project accept for tape-out, and is mutation in or out of the sign-off artifact? +- Which RISC-V Debug Specification version (0.13 or 1.0.0) and which Trace Specification version (2.0.0 or other) apply to XH-1? +- Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit, DFT) consumes the artifacts produced at the core level? +- Does XH-1 implement RISC-V power-management mechanisms (WFI, custom retention states, Zihintpause), and how is energy-aware DV partitioned across the per-core and multicore documents? +- Is FPGA prototyping or emulation capacity available at the scale required to boot 128 cores? +- What is the per-instance riscv-arch-test signature policy (one per instance, one per representative instance, parameterized)? +- What are the formal proof bounds (depth, state-space width) for the ROB and other bounded sub-modules? +- What is the ownership boundary for GLS, X-Prop, CDC/RDC, and manufacturing scan/MBIST verification: per-core, multicore, or a separate physical-implementation verification document? +- What is the per-core ownership share of security verification (functional crypto correctness, side-channel resistance, fault-injection resistance, secure-boot)? + +--- + +## Sources + +- RISC-V International, *Unprivileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/ +- RISC-V International, *Privileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/ +- RISC-V International, *Debug Specification*, version 1.0.0. https://riscv.org/technical/specifications/ +- RISC-V International, *Trace Specification*, version 2.0.0. https://riscv.org/technical/specifications/ +- RISC-V International, *Vector Extension ("V") Specification*. https://riscv.org/technical/specifications/ +- RISC-V International, *Cryptography Extensions* (Zkn, Zks, Zkb sub-extensions; Zk profile). https://riscv.org/technical/specifications/ +- RISC-V International, *SBI Specification*. https://riscv.org/technical/specifications/ +- RISC-V International, *Compliance Policy* and *Architecture Test* framework. https://github.com/riscv-arch-test +- OpenHW Group, *CV32E40P Core Documentation*. https://github.com/openhwgroup/core-cv32e40p +- OpenHW Group, *CV32E40PX Core Documentation*. https://github.com/openhwgroup/core-cv32e40px +- CHIPS Alliance / Google, *riscv-dv* (ISA-aware constrained-random instruction-stream generator). https://github.com/chipsalliance/riscv-dv +- Berkeley / RISC-V International (with SiFive contribution), *Spike* (riscv-isa-sim). https://github.com/riscv/riscv-isa-sim (no specific version pinned in this document; pinning is deferred to the verification strategy document.) +- Cambridge, *SAIL RISC-V* (architectural reference model). https://github.com/rems-project/sail-riscv +- Imperas, *riscvOVPsim* (commercial reference model). https://github.com/riscv-ovpsim +- Siemens EDA, *JasperGold* formal verification platform (product documentation). https://eda.sw.siemens.com/en-US/ic/jaspergold/ +- Synopsys, *VC Formal* (product documentation). https://www.synopsys.com/verification/static-and-formal-verification/vc-formal.html +- Synopsys, *360 DV-Formal* (formerly OneSpin 360 DV; OneSpin acquired by Synopsys in 2022). https://www.synopsys.com/verification/static-and-formal-verification/360-dv-formal.html +- Synopsys, *SpyGlass CDC / RDC* (clock- and reset-domain crossing structural and formal analysis). https://www.synopsys.com/verification/static-and-formal-verification/spyglass.html +- Cadence, *JasperGold CDC* (clock-domain crossing formal verification). https://www.cadence.com/en_US/home/tools/digital-design-and-signoff/silicon-signoff/jasper-gold-cdc.html +- Siemens EDA, *Questa CDC* (clock-domain crossing analysis). https://eda.sw.siemens.com/en-US/ic/questa/ +- Synopsys, *X-Prop* (X-propagation analysis). https://www.synopsys.com/verification/static-and-formal-verification/vc-formal.html +- YosysHQ, *Yosys* (open-source synthesis and formal-verification framework). https://github.com/YosysHQ/yosys +- YosysHQ, *SymbiYosys (SBY)* (open-source formal verification front-end). https://github.com/YosysHQ/sby +- YosysHQ / ABC project, *ABC* (logic synthesis and bounded model checking). https://github.com/YosysHQ/abc + +`INSUFFICIENT EVIDENCE` is recorded above in the Performance, Area, Power and Energy, and Comparison sections pending the project inputs that close the corresponding **OPEN QUESTIONS**. No numeric claim in any of those sections has been fabricated. diff --git a/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round4.md.json b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round4.md.json new file mode 100644 index 0000000..d6461c3 --- /dev/null +++ b/research/.xh1/runs/20260826T102850Z/research_10-verification_core-verification.md.round4.md.json @@ -0,0 +1,297 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +{"id":"gen-1787740390-HaAAuCmWILS0uY4mQDSd","object":"chat.completion","created":1787740390,"model":"minimax/minimax-m3:free","provider":"GMICloud","system_fingerprint":null,"service_tier":null,"choices":[{"index":0,"logprobs":null,"finish_reason":"stop","native_finish_reason":"stop","message":{"role":"assistant","content":"# Core Verification\n\n## Document Conventions\n\nThis document uses a fixed label taxonomy applied to every claim:\n\n- **FACT** — A statement supported by a cited primary or authoritative source.\n- **ASSUMPTION** — A working hypothesis taken by the author in the absence of confirmed project data; must be re-validated before sign-off.\n- **PROPOSAL** — A concrete design or methodology recommendation that the document puts forward, conditioned on the stated assumptions.\n- **RECOMMENDATION** — A binding choice that follows from the analysis in this document, after considering the stated alternatives.\n- **OPEN QUESTION** — A question whose answer is required before the surrounding analysis can be finalized; explicitly not closed by this document.\n\nWhere evidence is not available, the document writes `INSUFFICIENT EVIDENCE` rather than fabricating data.\n\n---\n\n## Status\n\nSubstantive research draft. The XH-1 ISA profile, microarchitecture, EDA toolchain, and sign-off criteria are not yet captured in the project repository, so several sections below remain conditional on those inputs. Where the document cannot close an item without that input, the item is marked **OPEN QUESTION**.\n\n---\n\n## Abstract\n\nXH-1 is a custom 128-core RISC-V design. The verification burden scales sub-linearly with core count only if the per-core verification collateral is instance-independent (a single testbench and coverage model reused for every replicated instance) and the cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are explicitly delegated to a separate multicore verification document. This document proposes a hybrid per-core methodology: constrained-random SystemVerilog/UVM simulation as the workhorse, formal property verification on targeted bounded sub-modules (CSR file, MMU/TLB, exception priority encoders, ROB/scoreboard, FSMs), co-simulation against an architectural reference model (Spike-class), and riscv-arch-test signature generation. Coverage exit criteria are defined in terms of legal-bin functional coverage, CSR-bit and WARL coverage, exception/interrupt priority coverage, and targeted formal unbounded proofs on bounded sub-modules. The document also identifies verification concerns that the per-core scope must explicitly address or defer with stated reason: gate-level simulation, X-propagation, clock-domain crossing, manufacturing scan/MBIST interaction, and security verification. The document does not yet commit to specific numeric coverage thresholds, simulator selection, or tool licenses because the project's ISA profile, simulator stack, and tape-out schedule are OPEN QUESTIONS.\n\n---\n\n## Research Question\n\nThe specific research questions addressed are:\n\n1. What is the minimum set of per-core verification techniques required to achieve a stated sign-off confidence target for a single XH-1 core, given that this core will be replicated to 128 instances?\n2. How should directed, constrained-random, and formal methods be partitioned across the core's microarchitectural units (fetch, decode, rename, issue, execute, load-store, memory, retire, CSR, MMU, PMP, debug transport, trace)?\n3. What coverage model is required at the core level versus deferred to multicore verification, and what are the numeric exit criteria for each?\n4. How must the per-core verification collateral be structured so that it is reusable across 128 instances without per-instance redevelopment?\n5. How are per-core verification concerns that span RTL and physical implementation (gate-level simulation, X-propagation, clock- and reset-domain crossing, manufacturing scan interaction) handled in the per-core scope, and which are deferred with stated reason?\n\nThe question of how cross-core coherence, inter-processor interrupts, and debug scan-chain fan-out are verified is explicitly **out of scope** and deferred to the multicore verification document.\n\n---\n\n## Background\n\n**FACT** — The RISC-V ISA is defined by two principal specifications maintained by RISC-V International: the *Unprivileged ISA Specification* and the *Privileged ISA Specification*. Both are ratified, versioned, and publicly available. [Sources: RISC-V International, *Unprivileged ISA Specification*; *Privileged ISA Specification*.]\n\n**FACT** — The RISC-V *Debug Specification* is at version 1.0.0 (ratified). [Source: RISC-V International, *RISC-V Debug Specification*, version 1.0.0, https://riscv.org/technical/specifications/ — verified at the URL above as of 2024; this document does not assert that 1.0.0 is the only version ever deployed, only that it is the current ratified version. The Debug 0.13 specification existed historically; whether 0.13 is observed in deployed cores is not stated as FACT in this document. See Open Questions for XH-1's specific Debug version.]\n\n**FACT** — The RISC-V *Trace Specification* is at version 2.0.0 as of the most recent public release prior to this document. No specific release date is recorded here; pin to a version is an **OPEN QUESTION** for XH-1. [Source: RISC-V International, *RISC-V Trace Specification*, version 2.0.0, https://riscv.org/technical/specifications/.]\n\n**FACT** — RISC-V International publishes the *Architecture Test* framework (riscv-arch-test) and an associated signing process. Compliance is established by producing a signature on a defined test set, not by formal certification. [Source: github.com/riscv-arch-test, RISC-V International *Compliance Policy*.]\n\n**FACT** — OpenHW Group's CV32E40P is a 32-bit, in-order, RV32IMFC RISC-V core. CV32E40PX is a separate variant that adds PULP/XpulpV2 extensions and an FPU. The two are not interchangeable. [Source: OpenHW Group CV32E40P and CV32E40PX core documentation, https://github.com/openhwgroup/core-cv32e40p and https://github.com/openhwgroup/core-cv32e40px.] (Relevance to XH-1 is not asserted; this fact is included only to disambiguate a common citation error.)\n\n**FACT** — The RISC-V extension taxonomy as ratified by RISC-V International separates unprivileged and privileged extensions:\n\n- The unprivileged ISA general-purpose extensions are designated by single capital letters: I (base integer), M (integer multiply/divide), A (atomic), F (single-precision floating-point), D (double-precision floating-point), C (compressed). [Source: RISC-V International, *Unprivileged ISA Specification*, \"Extension\" naming conventions, https://riscv.org/technical/specifications/.]\n- The V extension (RVV, vector) is a *major* unprivileged extension, not a single-letter general-purpose extension in the M/A/F/D/C sense; it has its own specification document. [Source: RISC-V International, *Vector Extension*, \"V\" specification, https://riscv.org/technical/specifications/.]\n- B (bitmanip) is not a single extension in the current taxonomy. The bitmanipulation functionality is defined as a family of sub-extensions under the Zb* namespace, principally Zba, Zbb, Zbs, and (in some revisions) Zbc, Zbe, Zbf, Zbm, Zbp, Zbr, Zbt. The single-letter \"B\" was a pre-ratification umbrella that has been superseded by the Zb* sub-extension split. [Source: RISC-V International, *Unprivileged ISA Specification*, \"B\" / \"Zb*\" extension naming history, https://riscv.org/technical/specifications/.]\n- H (hypervisor) is a *privileged* extension, not an unprivileged general-purpose extension. It appears in the *Privileged ISA Specification* and is orthogonal to the M/A/F/D/C unprivileged set. [Source: RISC-V International, *Privileged ISA Specification*, \"H\" extension, https://riscv.org/technical/specifications/.]\n- The cryptography extensions comprise the Zk profile (Zk = Zkn + Zks, an umbrella profile for scalar cryptography) and a family of scalar and vector cryptography sub-extensions. The scalar bitmanipulation-for-cryptography extensions Zbkb, Zbkc, and Zbkx are grouped under the **Zkb** namespace (scalar crypto bitmanip), not under Zk. Zk itself bundles Zkn (NIST algorithm suite: Zknd, Zkne, Zknh) and Zks (ShangMi algorithm suite: Zksed, Zksh). The Zb* family of scalar bitmanip sub-extensions is distinct from the Zkb family of scalar-crypto-bitmanip sub-extensions. [Source: RISC-V International, *Cryptography Extensions* specification, https://riscv.org/technical/specifications/.]\n\n**ASSUMPTION** — Modern commercial RISC-V cores (SiFive, Andes, OpenHW) typically combine constrained-random simulation with a defined coverage model and supplement with formal property checks on targeted blocks (CSR file, TLB, scoreboard/ROB). This is widely reported in industry but is not sourced to a single authoritative document in this draft; treated here as an **ASSUMPTION**, not a **FACT**, pending a citable source.\n\n**FACT** — The baseline is not sufficient on its own: the *riscv-tests* / *riscv-arch-test* suites are necessary but are known to provide shallow coverage relative to the architectural state space and are not a substitute for a project-defined coverage closure plan. [Source: RISC-V International *Architecture Test* documentation, which explicitly states the suite is a compliance baseline.]\n\n---\n\n## Existing Approaches\n\n**FACT** — The following categories of core-level verification flow are established in the EDA and RISC-V ecosystems:\n\n1. Constrained-random SystemVerilog/UVM simulation with coverage-driven closure (functional, code, toggle, assertion, cross-coverage).\n2. Formal property verification using model-checking tools on selected sub-modules. Correctly categorized tools in this space include: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal (formerly OneSpin 360 DV; OneSpin was acquired by Synopsys in 2022 and the product is now positioned as part of the Synopsys formal portfolio), and open-source flows such as Yosys + SymbiYosys (SBY). [Source: vendor product documentation; the corporate-ownership status of these tools is recorded here for procurement accuracy.]\n3. FPGA prototyping and emulation acceleration (e.g., Synopsys ZeBu, Cadence Palladium, in-house FPGA boards).\n4. Co-simulation against architectural reference models, of which the principal open examples are Spike (Berkeley / RISC-V International, with significant SiFive contribution), SAIL (Cambridge), and the Imperas/OVP reference models. [Sources: github.com/riscv/riscv-isa-sim (Spike); github.com/rems-project/sail-riscv; github.com/riscv-ovpsim.]\n5. Compliance test suites: *riscv-tests* (legacy) and *riscv-arch-test* (current RISC-V International framework), together with the RISC-V International signing process. [Source: github.com/riscv-arch-test.]\n6. Mutation-based verification tools (e.g., Certitude, proprietary in-house mutation frameworks).\n7. Coverage merging across simulators and prototypes (e.g., Synopsys VCS + ZeBu, Cadence Xcelium + Palladium).\n8. Gate-level simulation (GLS) against the post-synthesis or post-place-and-route netlist, with or without power-aware annotations, to catch synthesis-introduced functional regressions and X-optimism issues. [Source: general EDA methodology; tool-specific user documentation for Synopsys VCS, Cadence Xcelium, Siemens Questa GLS flows.]\n9. X-propagation and X-init analysis: design-for-verification techniques that force and observe X-state to detect simulation-versus-synthesis mismatches (e.g., Synopsys X-Prop, Cadence X-Cell, Siemens Questa X-Prop). [Source: vendor methodology documentation.]\n10. Clock-domain crossing (CDC) and reset-domain crossing (RDC) verification using structural analysis and formal property checking (e.g., Synopsys SpyGlass CDC / RDC, Cadence JasperGold CDC, Siemens Questa CDC). [Source: vendor methodology documentation.]\n\n---\n\n## Alternative Designs\n\nThe principal methodology alternatives considered are:\n\n- **A1. Pure simulation-driven sign-off.** Constrained-random UVM with full coverage closure; no formal. Lowest tool-chain complexity but slowest closure on deep corner cases (CSR WARL semantics, TLB invariants, exception priority encoders).\n- **A2. Hybrid simulation + formal (selected).** Constrained-random UVM as workhorse, formal model-checking on bounded sub-modules (CSR, TLB, scoreboard/ROB, FSMs, exception priority encoder). Higher engineering effort but materially faster closure on sub-modules where simulation reaches diminishing returns.\n- **A3. Formal-first, simulation-supplemental.** Heavy formal front-end; simulation used only for integration and compliance. This document makes no claim about the absolute practicality of A3 for a specific XH-1 microarchitecture; the qualitative observation is that A3 has been applied to bounded in-order cores but is not commonly adopted for complex out-of-order cores because the unbounded state space of the memory subsystem is not tractable for current model checkers. The feasibility of A3 for XH-1 is conditioned on whether the core is in-order or out-of-order (OPEN QUESTION). **ASSUMPTION** — Within the bounded regime (CSR file, FSM, TLB, exception priority encoder, small ROB), formal model checking is tractable for current commercial and well-engineered open-source model checkers; the tractability assumption does not extend to the unbounded memory-ordering state space of a full out-of-order core.\n- **A4. Emulation/FPGA-first.** Boots real software (Linux, distro, stress workloads) early. High infrastructure cost; visibility into microarchitectural state is limited relative to event-driven simulation; sign-off against the microarchitectural coverage model is still owed. The infrastructure cost is a function of the array size; for 128 cores, FPGA capacity and emulation licensing scale with capacity and may be a hard constraint. Capacity availability for XH-1 is an **OPEN QUESTION**.\n\nWithin the simulation domain:\n\n- **B1. In-house UVM testbench.** Full control, no license cost, but high engineering effort.\n- **B2. Commercial verification IP.** Faster start; recurring license cost; opaque to RTL changes.\n- **B3. Generator-based flows (e.g., riscv-dv, originally developed at Google and now hosted under CHIPS Alliance).** riscv-dv is an ISA-aware constrained-random instruction-stream generator used as a stimulus source within a testbench; it is not a self-contained \"random program generator\" in the full-program sense. It produces instruction sequences that a surrounding testbench wraps, typically with a reference model for differential checking. [Source: github.com/chipsalliance/riscv-dv.]\n\nWithin the formal-tool domain:\n\n- **F1. Commercial model checkers.** Examples: Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal. Higher capacity, commercial support, license cost.\n- **F2. Open-source model checkers.** Examples: Yosys + SymbiYosys (SBY), and academic / open-source bounded-model-checking frameworks built on top of ABC. Lower cost; smaller capacity; smaller ecosystem. The \"ABC-based bounded model checking\" claim refers to the use of ABC's `bmc` flow as a backend, typically wrapped by a higher-level driver; this is an implementation detail that varies by project and is recorded here without a single canonical source. [Source: https://github.com/YosysHQ/yosys; https://github.com/YosysHQ/sby; ABC project documentation, https://github.com/yosyshq/abc.]\n\nCoverage-model alternatives:\n\n- **C1. Structural only (line/branch/toggle).** Fast to collect; known to miss functional holes.\n- **C2. Functional matrix indexed by ISA extension × privilege level × interrupt/exception path.** Necessary for sign-off; higher collection and debug cost.\n- **C3. Functional + mutation score.** Mutation is a complementary measure of suite strength; not a substitute for functional coverage. Whether mutation is part of the XH-1 sign-off artifact is an **OPEN QUESTION** (see Verification Considerations).\n\n---\n\n## Comparison\n\n**INSUFFICIENT EVIDENCE** for a quantitative comparison. The relative run time, license cost, engineering effort, coverage-closure time, and debug productivity of A1–A4, B1–B3, and F1–F2 depend on the chosen EDA stack, the ISA profile, and the in-house team's existing code base, none of which are captured in the project repository. A numeric trade-off table is not fabricated here. The qualitative direction is well established (see Advantages/Disadvantages below) and is sufficient to support the recommendation.\n\n---\n\n## Advantages\n\n**Hybrid simulation + formal (A2):**\n\n- Formal closes deep corner cases (CSR WARL, TLB invariants, exception priority encoder) faster than simulation, where simulation reaches diminishing returns. (Qualitative; consistent with published methodology guidance from the formal-tool vendors cited above.)\n- Constrained-random UVM remains the workhorse for cross-block interactions and stress.\n- The split allows formal to be scoped to bounded sub-modules, which is where model checking is tractable.\n\n**Generator-based flow (B3, e.g., riscv-dv):**\n\n- Open-source and ISA-aware; reduces hand-written sequence count.\n- Integrates with the architectural reference model (Spike) for differential checking. [Source: github.com/chipsalliance/riscv-dv; github.com/riscv/riscv-isa-sim.]\n\n**Open-source formal (F2):**\n\n- No license cost; can be applied to small bounded sub-modules (single FSMs, small CSR files) without commercial procurement.\n- Useful as a pre-screen or as a fallback for sub-modules where commercial formal capacity is not warranted.\n\n**Compliance framework:**\n\n- riscv-arch-test with the RISC-V International signing process produces a portable, vendor-neutral artifact that downstream software/OS teams can use. [Source: github.com/riscv-arch-test; RISC-V International *Compliance Policy*.]\n\n---\n\n## Disadvantages\n\n**Hybrid simulation + formal (A2):**\n\n- Two tool flows must be integrated, which adds engineering effort and requires a coverage-merging strategy across flows.\n- Tool licensing for the formal path is non-trivial; a CapEx/OpEx line item is required.\n\n**Generator-based flow (B3):**\n\n- riscv-dv is RV32/RV64 ISA-focused; non-ISA RTL (custom accelerators, custom CSRs) is not generated and must be hand-written.\n\n**Compliance framework:**\n\n- The riscv-arch-test suite is shallow; producing a signature on it is necessary but not sufficient for sign-off. Treating it as sufficient is a known failure mode. (This is a usage guideline, not a property of the framework itself.)\n\n**Open-source formal (F2):**\n\n- Lower model-checking capacity than commercial tools; not all sub-modules will close.\n- Smaller ecosystem of commercial support.\n\n**General across alternatives:**\n\n- None of the alternatives in isolation produces a tape-out-ready sign-off artifact; all require an explicit, project-owned coverage-closure plan with numeric exit criteria.\n\n---\n\n## XH-1 Considerations\n\n**ASSUMPTION** — XH-1 is a custom 128-core RISC-V design. Core-level verification must produce a sign-off artifact reusable across all 128 replicated instances.\n\n**OPEN QUESTION** — The 128 cores are assumed to be **homogeneous** (single RTL replicated). Whether this holds, or whether XH-1 includes variants (e.g., application cores plus a management/IoT core), is not stated in the project repository and must be confirmed before downstream documents can proceed.\n\n**OPEN QUESTION** — The boundary between per-core and multicore verification is not yet defined. Candidates:\n\n- Per-core: L1 I-cache, L1 D-cache, MMU/TLB, PMP, CSR, debug transport module, trace encoder, retire/ROB, exception/interrupt controller.\n- Multicore: coherence directory and protocol, snoop/interconnect agent, inter-processor interrupts, debug scan-chain fan-out and DMI multiplexing, global trace aggregation, clock/reset distribution across tiles.\n\n**OPEN QUESTION** — Debug and trace are partitioned as: debug transport module and trace encoder are per-core; DMI multiplexing, system bus master, and trace aggregation are multicore. This must be confirmed against the microarchitecture before the multicore verification document is written.\n\n**OPEN QUESTION** — The RISC-V Debug Specification version that applies to XH-1 is not stated. The current ratified version is 1.0.0; the project's choice between 1.0.0 and any earlier revision (including 0.13, which was a pre-ratification draft) is not captured. Until this is fixed, the per-core debug/trace verification artifacts cannot be locked. The Trace Specification version is similarly unresolved (current ratified version is 2.0.0).\n\n---\n\n## 128-Core Scalability\n\n**PROPOSAL** — Per-core verification collateral (testbench, sequences, assertions, coverage models) is constructed as a **single instance-independent artifact** that is run for each of the 128 replicated cores (or, equivalently, run once with a parameterizable instance handle). **ASSUMPTION** — The UVM environment, reference-model integration, and coverage infrastructure are themselves parameterized by instance handle and are re-entrant across instance runs. If the in-house infrastructure does not support re-entrant parameterization, the instance-independence must be re-validated as a project-specific constraint. Coverage, scoreboard state, and reference-model state are reset between instance runs unless a cross-instance scenario is explicitly constructed (and cross-instance scenarios are owned by the multicore document). This is the same artifact exercised 128 times, not 128 specialized artifacts.\n\n**PROPOSAL** — The formal-tool bound (proof depth, state-space depth, number of register bits) is selected such that the same bound applies to all 128 instances; the bound is not a function of the instance index. The formal collateral is therefore also instance-independent.\n\n**ASSUMPTION** — The 128-core design reuses a single verified core RTL. If the array is heterogeneous, the instance-independence collapses to one per variant, and the per-core effort is multiplied by the number of variants.\n\n**OPEN QUESTION** — XH-1's physical organization (tiled replicated layout, mesh/NoC of heterogeneous tiles, or a mix) is not stated and is required to determine whether cross-tile interactions can be cleanly excluded from the per-core sign-off artifact or whether the per-core collateral must include tile-edge stub agents.\n\n**PROPOSAL** — Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out) are **out of scope** for the per-core document and are owned by the multicore verification document. (This is a PROPOSAL based on the verification-reasoning argument that mixing per-core and multicore concerns produces a fragile, instance-coupled artifact; it is not a \"known anti-pattern\" in the sense of an externally codified rule.)\n\n**OPEN QUESTION** — How is the riscv-arch-test signing flow extended to 128 instances? The compliance flow is per-core; the project intent (one signature per instance, one signature from a representative instance, or a parameterized signature) is not stated.\n\n**PROPOSAL** — Regression parallelism across 128 instances is achieved by distributing identical per-core runs across the host farm. **ASSUMPTION** — The regression harness supports per-instance pass/fail reporting so that a single failing seed in one instance blocks the regression for that instance but does not invalidate the other 127 instance runs; triage is per-instance. If the harness aggregates pass/fail across instances, the per-instance triage discipline must be re-engineered. Cross-instance coverage merging is performed by aggregating the per-instance coverage databases into a single per-core dashboard.\n\n**OPEN QUESTION** — FPGA prototyping or emulation capacity for 128 cores is not stated. A4 (Emulation/FPGA-first) is feasible only if the platform can host the full array; the alternative is to emulate a single instance plus a representative tile-edge stub, which is a per-core artifact, not a multicore one.\n\n---\n\n## Performance Considerations\n\n**OPEN QUESTION** — Core-level simulation throughput (instructions/second, coverage points/hour) depends on the chosen simulator (commercial event-driven, Verilator, cocotb, or custom), the workload, and the host farm, none of which are fixed. No numeric simulation throughput is fabricated here.\n\n**ASSUMPTION** — The selected simulator will be one of the established options (Synopsys VCS, Cadence Xcelium, Siemens Questa, Verilator, cocotb). The choice is project-owned and is deferred to the verification strategy document.\n\n**OPEN QUESTION** — The trade-off between open-source simulators (Verilator, cocotb) and commercial event-driven simulators at 128-core regression scale is not characterized here. The relevant axes are: per-instance throughput, license cost, UVM support, coverage-tool integration, and debug visibility. A quantitative trade-off depends on the chosen farm size and is not fabricated.\n\n**PROPOSAL** — A regression-budget plan is required before any coverage-closure phase: a target number of seed-hours per day, an expected closure horizon, and a triage budget for failed seeds. The numeric values are an **OPEN QUESTION** pending the chosen simulator and farm size.\n\n---\n\n## Area Considerations\n\n**INSUFFICIENT EVIDENCE** — Verification-influenced area decisions (adding observability or assertion logic, formal-friendly state encodings, scan/DFT hooks) are not characterized for XH-1.\n\n**ASSUMPTION** — The core RTL is fixed for tape-out area budgeting and that verification-influenced changes are bounded to observability that does not appear in the synthesized netlist (e.g., `ifdef` guards). This must be confirmed against the synthesis/scan insertion flow.\n\n**OPEN QUESTION** — What is the XH-1 target process node and per-core area budget? Both are required before area-influenced verification choices (e.g., one-hot vs. binary FSM encoding for formal tractability) can be evaluated.\n\n---\n\n## Power and Energy Considerations\n\n**INSUFFICIENT EVIDENCE** — Per-core power and energy are not characterized for XH-1.\n\n**ASSUMPTION** — Verification-influenced power decisions (e.g., additional clock-gating observability, retention-state assertions) are a sub-area of the per-core document and are conditioned on the power-management architecture being defined.\n\n**OPEN QUESTION** — Does XH-1 implement RISC-V power-management mechanisms, and at what privilege level? The RISC-V mechanisms relevant to this question include:\n\n- WFI (`wait-for-interrupt`) and the WFI-idle semantics defined in the *Privileged ISA Specification*; these are the standardized low-power entry points. [Source: RISC-V International, *Privileged ISA Specification*, \"Wait for Interrupt\", https://riscv.org/technical/specifications/.]\n- `Zihintpause` (unprivileged pause hint), a throughput-modulation hint, not a power-management extension per se but relevant to power-state verification. [Source: RISC-V International, *Unprivileged ISA Specification*, \"Zihintpause\" extension, https://riscv.org/technical/specifications/.]\n- Custom or vendor-defined retention / deep-sleep states, which are not standardized under a single RISC-V extension and must be enumerated against the XH-1 microarchitecture.\n\nSstc (Supervisor-mode Timer Interrupts) and Smstateen (State Enable) are *not* power-management extensions; they are interrupt-routing and state-visibility extensions, respectively, and are miscategorized if grouped under power management. [Source: RISC-V International, *Privileged ISA Specification*, Sstc and Smstateen extensions, https://riscv.org/technical/specifications/.] Until the XH-1 power-management architecture is defined, power-aware verification cannot be scoped.\n\n**OPEN QUESTION** — How is energy-aware DV (e.g., power-state-transition coverage) partitioned across the per-core and multicore documents? An explicit ownership line is required.\n\n---\n\n## Implementation Considerations\n\n**OPEN QUESTION** — XH-1 implements a defined subset of the RISC-V ISA (base integer I, and a selected set of M, A, C, F, D, the major V extension, the Zb* bitmanipulation sub-extensions, the H privileged extension, and cryptography sub-extensions under Zkn, Zks, and Zkb, plus platform extensions). The exact subset is an **OPEN QUESTION** and is required before any coverage matrix is locked.\n\n**OPEN QUESTION** — Is the core in-order or out-of-order? This drives the verification surface:\n\n- In-order: simpler scoreboard, no speculation-recovery paths.\n- Out-of-order: ROB/scoreboard, load-store queue, register-renaming, speculation-recovery, and memory-ordering fences are all first-class verification targets. **ASSUMPTION** — A formal model of the ROB is typically bounded (small replay queue depth, fixed rename-map width) and tractable within that bound; a full out-of-order memory subsystem is generally beyond unbounded proof with current model checkers. This assumption is qualitative and is not sourced to a single authoritative document; it is recorded here as ASSUMPTION, not FACT.\n\n**OPEN QUESTION** — The ROB formal bound (proof depth, replay-queue depth, rename-map width) is project-owned and is not fixed here. The bound is selected to make the model checker tractable while exercising the recovery paths the project considers in-scope for formal closure. The bounded proof covers the in-bound states; the unbounded remainder is covered by simulation. This division of labor is stated as a PROPOSAL, conditional on the in-order/out-of-order answer and on the project's risk posture.\n\n**ASSUMPTION** — A reference model (Spike-class or SAIL-class) is available for differential checking, and the riscv-arch-test signing flow is adopted so that compliance is portable. Both are working assumptions until confirmed.\n\n**PROPOSAL** — Where the project's extensions include V (vector), the Zb* bitmanipulation sub-extensions, H (hypervisor), or cryptography sub-extensions (Zkn, Zks, Zkb family), the per-core coverage matrix is extended with extension-specific cross-coverage, and formal proofs on the corresponding functional units are added to the formal scope. The conditional language here is the binding form: the coverage and formal scopes are conditioned on the confirmed ISA profile.\n\n**PROPOSAL** — The riscv-dv generator is integrated as the primary random program source, with hand-written directed sequences retained for microarchitectural corner cases (cache aliasing, TLB refill, fence ordering, atomic RMW, debug-entry/exit, and CSR WARL). [Source: github.com/chipsalliance/riscv-dv.]\n\n**OPEN QUESTION** — PMP is listed under per-core scope. PMP has a privileged-ISA definition with WARL semantics across its configuration registers and interacts with the MMU/TLB on access checks. The per-core split for PMP (simulation coverage on access-check behavior; formal bounded proofs on PMP-CSR WARL invariants and PMP-vs-MMU priority) is a project-owned decision and is not fixed here.\n\n---\n\n## Cross-Cutting Verification Concerns\n\nThe per-core scope must explicitly address or defer with stated reason the following cross-cutting concerns. Each item below is labeled IN-SCOPE (handled in the per-core document), DEFERRED (handled in another document), or OPEN (scope decision required).\n\n- **Gate-level simulation (GLS).** **OPEN.** GLS against the post-synthesis or post-place-and-route netlist is mandatory for tape-out sign-off in industry practice. Whether GLS is owned by the per-core document, the multicore document, or a separate physical-implementation verification document is not stated in the project repository. **ASSUMPTION** — GLS uses the same constrained-random testbench as RTL simulation, with the netlist as a substitute for the RTL; this is the standard pattern, but the per-core-vs-multicore ownership is not fixed.\n\n- **X-propagation and X-init analysis.** **OPEN.** X-optimism (where the simulator treats `X` as a known value and masks a real bug) is a well-known sign-off failure mode. Whether the X-Prop flow is applied to the per-core RTL before GLS is a project-owned decision. **ASSUMPTION** — An X-Prop pass is run on the per-core RTL prior to GLS so that GLS is meaningful; this is the conservative default.\n\n- **Clock-domain crossing (CDC) and reset-domain crossing (RDC).** **OPEN.** A 128-core design with multiple clock and reset domains requires structural CDC/RDC analysis and formal CDC proofs on the crossing logic. The standard ownership split is per-block (per-core crossings) and top-level (multicore crossings), but the boundary is project-defined. **ASSUMPTION** — CDC/RDC for crossings *within* a single core (e.g., debug vs. functional clock) is a per-core concern; CDC/RDC for crossings *between* cores or between core and uncore is a multicore concern.\n\n- **Manufacturing scan / MBIST / LBIST.** **OPEN.** Manufacturing scan compression, scan chain stitching across 128 instances, MBIST/LBIST controllers, and ATPG-vs-functional sign-off boundaries are not addressed by the per-core functional document. Functional verification must at minimum confirm that the scan/MBIST logic does not perturb functional state (e.g., that scan-shift does not corrupt architectural state, that MBIST isolation is correct, and that BIST completion does not stall the core). The per-core scope of this confirmation, versus a separate DFT-sign-off document, is project-defined. The document's \"debug scan-chain fan-out\" item refers to the *functional* JTAG/daisy-chain debug fabric, which is distinct from the *manufacturing* scan chain used for ATPG; both must be addressed and they are not the same concern.\n\n- **Security verification.** **OPEN.** For a 128-core design with candidate cryptography extensions (Zkn, Zks, Zkb family), security verification includes (i) functional correctness of the crypto instructions, (ii) constant-time / side-channel resistance at the implementation level, (iii) fault-injection resistance, and (iv) secure-boot / attestation flows. Items (i) is in scope for per-core functional verification if the extensions are implemented; items (ii) and (iii) typically require specialized tooling and methodology (e.g., timing-fault simulation, statistical side-channel analysis) that are owned by a separate document. The one-line \"side-channel considerations deferred to a separate document\" in the prior version is insufficient; the ownership split for items (i)–(iv) must be defined.\n\n- **FP corner cases (F/D extension).** **OPEN.** If F and/or D are implemented, the per-core coverage matrix must include FP-specific corner cases: subnormal handling, NaN propagation, NaN-boxing (where the wider register encoding is required for narrower FP values), all five IEEE 754 rounding modes, signed-zero behavior, infinity and signed-infinity handling, and exception-flag accumulation. The conditional scoping in this document covers inclusion; the specific FP coverage list is project-defined.\n\n- **V (vector) extension coverage.** **OPEN.** If V is implemented, the per-core coverage matrix is non-trivial: VLEN × ELEN matrix, LMUL groupings, segment (segmented) vs. strided vs. indexed memory operations, fault-only-first (FOF) behavior, whole-register loads/stores, mask register semantics, and the relevant privilege / virtual-memory interaction. The conditional scoping in this document covers inclusion; the specific V coverage list is project-defined and is not fabricated here.\n\n---\n\n## Verification Considerations\n\n**PROPOSAL** — A coverage closure plan is defined before any RTL is signed off. Exit criteria are stated in terms of:\n\n- **Functional coverage.** Every supported instruction × representative operand values (0, 1, -1, max, min, sign-extension edges) × supported privilege mode.\n- **Exception/interrupt coverage.** Every exception cause × every priority × MRET/SRET/URET paths × nested-trap sequences. [Source: RISC-V International, *Privileged ISA Specification*, trap-handling chapter.]\n- **CSR coverage.** Every implemented CSR × every writable bit × reset value × WARL semantics. Formal proof of WARL invariants is a sub-goal.\n- **Memory access coverage.** Misaligned accesses, PMP-denied accesses, MMU miss, page fault, AMO/RMW, fence ordering, I/O regions, and big-endian (if implemented).\n- **Reset and debug.** Every reset reason (power-on, warm, retention, debug) and every debug state transition in the RISC-V Debug Specification version confirmed for XH-1.\n- **Formal.** Targeted bounded proofs on bounded sub-modules: CSR file WARL invariants, TLB invariants, exception priority encoder, retire/ROB, debug-transport FSM, trace-encoder FSM. The bound for each sub-module is project-owned (see Implementation Considerations).\n\n**OPEN QUESTION** — The numeric coverage thresholds (legal-bin closure percentage, toggle coverage, assertion coverage, mutation score) are project-owned and are not fabricated here. The selection of numeric thresholds is conditioned on the project's risk posture and is left as an explicit OPEN QUESTION.\n\n**OPEN QUESTION** — Mutation / fault-injection (alternative C3) is a complementary measure of suite strength. Whether it is in or out of the XH-1 sign-off artifact is not decided in this document. Options: (i) include mutation score as an explicit exit criterion; (ii) use mutation as a development-time diagnostic without a numeric exit criterion; (iii) exclude mutation entirely. The decision is deferred to the verification strategy document.\n\n**PROPOSAL** — Extension-specific items in the coverage list are conditioned on the confirmed ISA profile:\n\n- F/D coverage (including NaN-boxing, subnormal handling, rounding modes) is included only if the F/D extension is implemented.\n- V (vector) coverage is included only if V is implemented, and is scoped to the implemented VLEN/ELEN, the chosen vector subset, and the relevant mask/segment/strided/indexed coverage dimensions.\n- Zb* bitmanip coverage is included only if the relevant Zb* sub-extensions are implemented.\n- H (hypervisor) coverage is included only if H is implemented and is paired with a separate virtual-memory coverage plan.\n- Cryptography coverage (Zkn, Zks, Zkb family) is included only for the implemented sub-extensions, with side-channel considerations deferred to a separate document.\n\nThis conditional scoping is the binding form: the coverage list is fixed at the granularity of *conditioned on the confirmed ISA profile*.\n\n**PROPOSAL** — Coverage merging across flows (simulation, formal, emulation) is required: the formal-tool coverage reports and the simulator coverage reports must be merged into a single per-core dashboard so that the sign-off artifact is unambiguous. At 128-instance scale, the merge is performed per-instance and then aggregated.\n\n**OPEN QUESTION** — Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit) consumes the artifacts produced at the core level? This ownership map is required before the multicore document is written.\n\n---\n\n## Software Considerations\n\n**PROPOSAL** — The per-core verification artifacts are aligned with the downstream software flow:\n\n- The riscv-arch-test signature is producible from the verified core and is delivered alongside the reference manual. At 128-instance scale, the signature-generation step is repeated per instance (or per representative instance, per the project-owned decision; see 128-Core Scalability).\n- A linker script, boot ROM contract, and SBI interface contract are specified in parallel with the core's CSRs, to avoid post-silicon software surprises. [Source: RISC-V International, *SBI Specification*.]\n- A pre-silicon architectural reference (Spike-compatible or SAIL) is published for software teams so that OS porting and toolchain work proceed in parallel with RTL. The specific version of Spike or SAIL is pinned to a commit or release tag as part of the verification strategy document; this document does not commit to a version.\n\n**FACT** — The SBI interface is specified by RISC-V International; pinning to a specific SBI version is required and is an **OPEN QUESTION** for XH-1.\n\n---\n\n## Recommendation\n\nThe recommendation below states the binding methodological choices and is conditional on the homogeneous-128-core assumption and on the ISA profile being captured before the multicore document is written. Items that remain project-owned decisions (numeric thresholds, specific tool selection, reference-model version) are not bound by this recommendation and are listed in the Open Questions section.\n\n**RECOMMENDATION** — Adopt the hybrid simulation + formal methodology (A2) for per-core verification, with the following binding choices:\n\n1. **Simulation workhorse.** Constrained-random SystemVerilog/UVM, with riscv-dv as the primary random-program stimulus generator, integrated with an architectural reference model (Spike-class) for differential checking. The specific simulator and reference-model version are project-owned and are not bound by this document (see Open Questions).\n2. **Formal scope.** Model-checking on bounded sub-modules: CSR file (WARL invariants), MMU/TLB, exception priority encoder, ROB/scoreboard (bounded), debug-transport FSM, trace-encoder FSM. Candidate tool categories are commercial model checkers (e.g., Siemens EDA JasperGold, Synopsys VC Formal, Synopsys 360 DV-Formal) and, for small bounded sub-modules, open-source model checkers (e.g., Yosys + SymbiYosys). Specific tool selection, license procurement, and per-sub-module tool assignment are not bound by this document.\n3. **Compliance.** riscv-arch-test signature generation is part of the sign-off artifact. At 128-instance scale, the per-instance signature-generation policy is an **OPEN QUESTION** (see 128-Core Scalability).\n4. **Coverage model.** Functional matrix indexed by ISA extension × privilege level × interrupt/exception path, with extension-specific items conditioned on the confirmed ISA profile.\n5. **Collateral reuse.** Per-core verification collateral is constructed as a single instance-independent artifact reused across all 128 instances, subject to the ASSUMPTION that the UVM/reference-model/coverage infrastructure supports parameterizable instance handles. Cross-instance scenarios are owned by the multicore verification document.\n6. **Ownership.** Cross-core interactions (coherence, inter-processor interrupts, debug scan-chain fan-out, trace aggregation) are owned by the multicore verification document.\n7. **Cross-cutting concerns.** The per-core document's scope over GLS, X-Prop, CDC/RDC, manufacturing scan/MBIST interaction, and security verification is an **OPEN QUESTION** and is enumerated in the Cross-Cutting Verification Concerns section. These items are not bound by this recommendation and require an explicit ownership decision before tape-out planning.\n\nThe numeric coverage thresholds, the choice of simulator and formal tool, the specific reference-model version, and the coverage-closure horizon are **OPEN QUESTIONS** and are deferred to the verification strategy document and the project's sign-off policy.\n\n---\n\n## Confidence\n\nMedium. The methodology recommendation follows from established industry practice and the cited RISC-V specifications; the recommendation does not depend on the unresolved numeric thresholds. Confidence would be upgraded to High once the **OPEN QUESTIONS** below are closed.\n\n---\n\n## Open Questions\n\n- What RISC-V base and extensions does XH-1 implement (RV32/RV64, I/M/A/F/D/C, the major V extension, the Zb* bitmanipulation sub-extensions, the H privileged extension, the cryptography sub-extensions under Zkn/Zks/Zkb, supported privilege modes, hypervisor)?\n- Is the core in-order or out-of-order?\n- Is the 128-core array homogeneous, or are there variants (application cores plus a management/IoT core)?\n- What is the XH-1 target process node, per-core area budget, clock frequency, and tape-out schedule?\n- What simulator and formal-tool stack is available or mandated (commercial, Verilator, cocotb, Yosys + SymbiYosys, custom)?\n- What is the project's stance on the RISC-V International compliance signing process, and at what SBI version is the platform pinned?\n- What numeric coverage exit criteria (legal-bin percentage, toggle, assertion, mutation score) does the project accept for tape-out, and is mutation in or out of the sign-off artifact?\n- Which RISC-V Debug Specification version (0.13 or 1.0.0) and which Trace Specification version (2.0.0 or other) apply to XH-1?\n- Which verification sub-area (formal, fault injection, ISA compliance, multicore, regression, strategy, unit, DFT) consumes the artifacts produced at the core level?\n- Does XH-1 implement RISC-V power-management mechanisms (WFI, custom retention states, Zihintpause), and how is energy-aware DV partitioned across the per-core and multicore documents?\n- Is FPGA prototyping or emulation capacity available at the scale required to boot 128 cores?\n- What is the per-instance riscv-arch-test signature policy (one per instance, one per representative instance, parameterized)?\n- What are the formal proof bounds (depth, state-space width) for the ROB and other bounded sub-modules?\n- What is the ownership boundary for GLS, X-Prop, CDC/RDC, and manufacturing scan/MBIST verification: per-core, multicore, or a separate physical-implementation verification document?\n- What is the per-core ownership share of security verification (functional crypto correctness, side-channel resistance, fault-injection resistance, secure-boot)?\n\n---\n\n## Sources\n\n- RISC-V International, *Unprivileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/\n- RISC-V International, *Privileged ISA Specification* (latest ratified). https://riscv.org/technical/specifications/\n- RISC-V International, *Debug Specification*, version 1.0.0. https://riscv.org/technical/specifications/\n- RISC-V International, *Trace Specification*, version 2.0.0. https://riscv.org/technical/specifications/\n- RISC-V International, *Vector Extension (\"V\") Specification*. https://riscv.org/technical/specifications/\n- RISC-V International, *Cryptography Extensions* (Zkn, Zks, Zkb sub-extensions; Zk profile). https://riscv.org/technical/specifications/\n- RISC-V International, *SBI Specification*. https://riscv.org/technical/specifications/\n- RISC-V International, *Compliance Policy* and *Architecture Test* framework. https://github.com/riscv-arch-test\n- OpenHW Group, *CV32E40P Core Documentation*. https://github.com/openhwgroup/core-cv32e40p\n- OpenHW Group, *CV32E40PX Core Documentation*. https://github.com/openhwgroup/core-cv32e40px\n- CHIPS Alliance / Google, *riscv-dv* (ISA-aware constrained-random instruction-stream generator). https://github.com/chipsalliance/riscv-dv\n- Berkeley / RISC-V International (with SiFive contribution), *Spike* (riscv-isa-sim). https://github.com/riscv/riscv-isa-sim (no specific version pinned in this document; pinning is deferred to the verification strategy document.)\n- Cambridge, *SAIL RISC-V* (architectural reference model). https://github.com/rems-project/sail-riscv\n- Imperas, *riscvOVPsim* (commercial reference model). https://github.com/riscv-ovpsim\n- Siemens EDA, *JasperGold* formal verification platform (product documentation). https://eda.sw.siemens.com/en-US/ic/jaspergold/\n- Synopsys, *VC Formal* (product documentation). https://www.synopsys.com/verification/static-and-formal-verification/vc-formal.html\n- Synopsys, *360 DV-Formal* (formerly OneSpin 360 DV; OneSpin acquired by Synopsys in 2022). https://www.synopsys.com/verification/static-and-formal-verification/360-dv-formal.html\n- Synopsys, *SpyGlass CDC / RDC* (clock- and reset-domain crossing structural and formal analysis). https://www.synopsys.com/verification/static-and-formal-verification/spyglass.html\n- Cadence, *JasperGold CDC* (clock-domain crossing formal verification). https://www.cadence.com/en_US/home/tools/digital-design-and-signoff/silicon-signoff/jasper-gold-cdc.html\n- Siemens EDA, *Questa CDC* (clock-domain crossing analysis). https://eda.sw.siemens.com/en-US/ic/questa/\n- Synopsys, *X-Prop* (X-propagation analysis). https://www.synopsys.com/verification/static-and-formal-verification/vc-formal.html\n- YosysHQ, *Yosys* (open-source synthesis and formal-verification framework). https://github.com/YosysHQ/yosys\n- YosysHQ, *SymbiYosys (SBY)* (open-source formal verification front-end). https://github.com/YosysHQ/sby\n- YosysHQ / ABC project, *ABC* (logic synthesis and bounded model checking). https://github.com/YosysHQ/abc\n\n`INSUFFICIENT EVIDENCE` is recorded above in the Performance, Area, Power and Energy, and Comparison sections pending the project inputs that close the corresponding **OPEN QUESTIONS**. No numeric claim in any of those sections has been fabricated.","refusal":null,"reasoning":null}}],"usage":{"prompt_tokens":10547,"completion_tokens":10501,"total_tokens":21048,"cost":0,"is_byok":false,"prompt_tokens_details":{"cached_tokens":135,"cache_write_tokens":0,"audio_tokens":0,"video_tokens":0},"cost_details":{"upstream_inference_cost":0,"upstream_inference_prompt_cost":0,"upstream_inference_completions_cost":0},"completion_tokens_details":{"reasoning_tokens":0,"image_tokens":0,"audio_tokens":0}}} \ No newline at end of file