mirror of
https://github.com/allexanderbergmns/xh1-research.git
synced 2026-08-26 16:07:01 +00:00
setup
This commit is contained in:
@@ -0,0 +1,117 @@
|
|||||||
|
[2026-08-25T17:58:32Z] Starting research: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T17:58:32Z] Running researcher.
|
||||||
|
[2026-08-25T17:58:32Z] API attempt 1 failed.
|
||||||
|
[2026-08-25T17:58:37Z] API attempt 2 failed.
|
||||||
|
[2026-08-25T17:59:55Z] Starting research: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T17:59:55Z] Running researcher.
|
||||||
|
[2026-08-25T17:59:56Z] API attempt 1 failed.
|
||||||
|
[2026-08-25T17:59:57Z] API attempt 2 failed.
|
||||||
|
[2026-08-25T18:00:09Z] Starting research: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:00:09Z] Running researcher.
|
||||||
|
[2026-08-25T18:00:10Z] API attempt 1 failed.
|
||||||
|
[2026-08-25T18:00:11Z] API attempt 2 failed.
|
||||||
|
[2026-08-25T18:00:15Z] API attempt 3 failed.
|
||||||
|
[2026-08-25T18:00:56Z] Starting research: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:00:56Z] Running researcher.
|
||||||
|
[2026-08-25T18:00:57Z] API attempt 1 failed.
|
||||||
|
[2026-08-25T18:00:58Z] API attempt 2 failed.
|
||||||
|
[2026-08-25T18:01:02Z] API attempt 3 failed.
|
||||||
|
[2026-08-25T18:01:11Z] Research agent failed.
|
||||||
|
[2026-08-25T18:01:38Z] Starting research: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:01:38Z] Running researcher.
|
||||||
|
[2026-08-25T18:01:38Z] API attempt 1 failed.
|
||||||
|
[2026-08-25T18:01:57Z] Starting research: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:01:57Z] Running researcher.
|
||||||
|
[2026-08-25T18:02:54Z] Starting research: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:02:54Z] Running researcher.
|
||||||
|
[2026-08-25T18:02:54Z] API attempt 1 failed.
|
||||||
|
[2026-08-25T18:02:55Z] API attempt 2 failed.
|
||||||
|
[2026-08-25T18:03:00Z] Starting research: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:03:00Z] Running researcher.
|
||||||
|
[2026-08-25T18:04:04Z] Research completed.
|
||||||
|
[2026-08-25T18:04:04Z] Running reviewer.
|
||||||
|
[2026-08-25T18:04:28Z] Research rejected by reviewer.
|
||||||
|
[2026-08-25T18:08:09Z] Started run: 20260825T180809Z
|
||||||
|
[2026-08-25T18:08:09Z] ==================================================
|
||||||
|
[2026-08-25T18:08:09Z] Researching: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:08:09Z] ==================================================
|
||||||
|
[2026-08-25T18:08:09Z] Research round 1/3
|
||||||
|
[2026-08-25T18:08:09Z] Running researcher.
|
||||||
|
[2026-08-25T18:09:00Z] Started run: 20260825T180900Z
|
||||||
|
[2026-08-25T18:09:00Z] ==================================================
|
||||||
|
[2026-08-25T18:09:00Z] Researching: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:09:00Z] ==================================================
|
||||||
|
[2026-08-25T18:09:00Z] Research round 1/3
|
||||||
|
[2026-08-25T18:09:00Z] Running researcher.
|
||||||
|
[2026-08-25T18:09:00Z] API request: model=minimax/minimax-m3:free attempt=1
|
||||||
|
[2026-08-25T18:09:56Z] API response received: 17248 bytes
|
||||||
|
[2026-08-25T18:09:56Z] Running reviewer.
|
||||||
|
[2026-08-25T18:09:56Z] API request: model=z-ai/glm-5.2:free attempt=1
|
||||||
|
[2026-08-25T18:09:57Z] API request failed.
|
||||||
|
[2026-08-25T18:09:58Z] API request: model=z-ai/glm-5.2:free attempt=2
|
||||||
|
[2026-08-25T18:09:58Z] API request failed.
|
||||||
|
[2026-08-25T18:10:02Z] API request: model=z-ai/glm-5.2:free attempt=3
|
||||||
|
[2026-08-25T18:10:03Z] API request failed.
|
||||||
|
[2026-08-25T18:11:53Z] Started run: 20260825T181153Z
|
||||||
|
[2026-08-25T18:11:53Z] ==================================================
|
||||||
|
[2026-08-25T18:11:53Z] Researching: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:11:53Z] ==================================================
|
||||||
|
[2026-08-25T18:11:53Z] Research round 1/3
|
||||||
|
[2026-08-25T18:11:53Z] Running researcher.
|
||||||
|
[2026-08-25T18:11:53Z] API request: model=minimax/minimax-m3:free attempt=1
|
||||||
|
[2026-08-25T18:13:01Z] API response received: 21756 bytes
|
||||||
|
[2026-08-25T18:13:01Z] Running reviewer.
|
||||||
|
[2026-08-25T18:13:01Z] API request: model=google/gemma-4-31b-it:free attempt=1
|
||||||
|
[2026-08-25T18:13:01Z] API request failed.
|
||||||
|
[2026-08-25T18:13:02Z] API request: model=google/gemma-4-31b-it:free attempt=2
|
||||||
|
[2026-08-25T18:13:02Z] API request failed.
|
||||||
|
[2026-08-25T18:13:06Z] API request: model=google/gemma-4-31b-it:free attempt=3
|
||||||
|
[2026-08-25T18:13:07Z] API request failed.
|
||||||
|
[2026-08-25T18:14:02Z] Started run: 20260825T181402Z
|
||||||
|
[2026-08-25T18:14:02Z] ==================================================
|
||||||
|
[2026-08-25T18:14:02Z] Researching: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:14:02Z] ==================================================
|
||||||
|
[2026-08-25T18:14:02Z] Research round 1/3
|
||||||
|
[2026-08-25T18:14:02Z] Running researcher.
|
||||||
|
[2026-08-25T18:14:02Z] API request: model=minimax/minimax-m3:free attempt=1
|
||||||
|
[2026-08-25T18:15:25Z] API response received: 26879 bytes
|
||||||
|
[2026-08-25T18:15:25Z] Running reviewer.
|
||||||
|
[2026-08-25T18:15:25Z] API request: model=poolside/laguna-s-2.1:free attempt=1
|
||||||
|
[2026-08-25T18:15:26Z] API request failed.
|
||||||
|
[2026-08-25T18:15:27Z] API request: model=poolside/laguna-s-2.1:free attempt=2
|
||||||
|
[2026-08-25T18:15:27Z] API request failed.
|
||||||
|
[2026-08-25T18:15:31Z] API request: model=poolside/laguna-s-2.1:free attempt=3
|
||||||
|
[2026-08-25T18:15:31Z] API request failed.
|
||||||
|
[2026-08-25T18:16:09Z] Started run: 20260825T181609Z
|
||||||
|
[2026-08-25T18:16:09Z] ==================================================
|
||||||
|
[2026-08-25T18:16:09Z] Researching: research/03-core-design/mul-div-unit.md
|
||||||
|
[2026-08-25T18:16:09Z] ==================================================
|
||||||
|
[2026-08-25T18:16:09Z] Research round 1/3
|
||||||
|
[2026-08-25T18:16:09Z] Running researcher.
|
||||||
|
[2026-08-25T18:16:09Z] API request: model=minimax/minimax-m3:free attempt=1
|
||||||
|
[2026-08-25T18:17:05Z] API response received: 18408 bytes
|
||||||
|
[2026-08-25T18:17:05Z] Running reviewer.
|
||||||
|
[2026-08-25T18:17:05Z] API request: model=minimax/minimax-m3:free attempt=1
|
||||||
|
[2026-08-25T18:17:41Z] API response received: 12530 bytes
|
||||||
|
[2026-08-25T18:17:41Z] Research rejected by reviewer.
|
||||||
|
[2026-08-25T18:17:41Z] Preparing revision round 2.
|
||||||
|
[2026-08-25T18:17:46Z] Research round 2/3
|
||||||
|
[2026-08-25T18:17:46Z] Running revision agent.
|
||||||
|
[2026-08-25T18:17:46Z] API request: model=minimax/minimax-m3:free attempt=1
|
||||||
|
[2026-08-25T18:18:49Z] API response received: 30663 bytes
|
||||||
|
[2026-08-25T18:18:49Z] Running reviewer.
|
||||||
|
[2026-08-25T18:18:49Z] API request: model=minimax/minimax-m3:free attempt=1
|
||||||
|
[2026-08-25T18:19:38Z] API response received: 17758 bytes
|
||||||
|
[2026-08-25T18:19:38Z] Research rejected by reviewer.
|
||||||
|
[2026-08-25T18:19:38Z] Preparing revision round 3.
|
||||||
|
[2026-08-25T18:19:43Z] Research round 3/3
|
||||||
|
[2026-08-25T18:19:43Z] Running revision agent.
|
||||||
|
[2026-08-25T18:19:43Z] API request: model=minimax/minimax-m3:free attempt=1
|
||||||
|
[2026-08-25T18:20:47Z] API response received: 38527 bytes
|
||||||
|
[2026-08-25T18:20:47Z] Running reviewer.
|
||||||
|
[2026-08-25T18:20:47Z] API request: model=minimax/minimax-m3:free attempt=1
|
||||||
|
[2026-08-25T18:21:21Z] API response received: 11009 bytes
|
||||||
|
[2026-08-25T18:21:21Z] Research rejected by reviewer.
|
||||||
|
[2026-08-25T18:21:21Z] Maximum research rounds reached.
|
||||||
|
[2026-08-25T18:21:21Z] Leaving original document unchanged.
|
||||||
|
[2026-08-25T18:21:21Z] Marking topic failed for this run.
|
||||||
@@ -0,0 +1,190 @@
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
@@ -0,0 +1,284 @@
|
|||||||
|
# MUL/DIV Unit
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
DRAFT — initial research document. No architectural decision has been made for XH-1.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
|
||||||
|
This document investigates the design of the multiplication and division unit (MUL/DIV) for a single core inside the XH-1 128-core RISC-V processor. It surveys existing approaches for integer multiplication and division, identifies alternative implementations, and analyzes trade-offs in performance, area, power, latency, verification complexity, and scalability across 128 cores. The document focuses on the base integer extensions (RV64I/M) and explicitly defers floating-point and vector multiply/accumulate topics, which belong to separate units.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
|
||||||
|
What is the most appropriate microarchitectural implementation of the MUL/DIV unit for a single XH-1 core, given that 128 identical cores will be instantiated on die, and given that the MUL/DIV unit must implement at minimum RV64M (MUL, MULH, MULHU, MULHSU, DIV, DIVU, REM, REMU)?
|
||||||
|
|
||||||
|
Sub-questions:
|
||||||
|
|
||||||
|
- Should multiplication be iterative (shift-and-add) or fully combinational (array/Wallace/Booth)?
|
||||||
|
- Should division use a restoring, non-restoring, SRT, or Newton–Raphson scheme?
|
||||||
|
- Should the unit be pipelined, multi-cycle, or variable-latency?
|
||||||
|
- How does the MUL/DIV unit interact with the surrounding pipeline (depth, bypass, in-order vs out-of-order issue)?
|
||||||
|
- How does the unit scale when replicated 128 times on die?
|
||||||
|
- Should fused MAC operations (MUL + ADD) or fused multiply-add be considered?
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
The RISC-V "M" extension specifies eight integer multiply/divide instructions (MUL, MULH, MULHSU, MULHU, DIV, DIVU, REM, REMU) on 64-bit values producing either 64-bit or 128-bit results. Division is defined to round toward zero and is required to complete even for overflow cases (e.g., `INT64_MIN / -1`), as specified in the RISC-V Unprivileged ISA.
|
||||||
|
|
||||||
|
Key properties of the workload that influence the design:
|
||||||
|
|
||||||
|
- **Result width asymmetry.** MUL produces a 128-bit result, but only the low 64 bits are written to `rd` for `MUL`. MULH-family instructions write the high 64 bits. The datapath therefore needs at least a 64×64→128 multiplier followed by a selector.
|
||||||
|
- **Sign handling.** Three signed/unsigned combinations (signed×signed, signed×unsigned, unsigned×unsigned) must be supported. Sign correction is required for MULH/MULHSU.
|
||||||
|
- **Division latency and throughput.** RISC-V does not require division to retire in a single cycle, but the ISA mandates deterministic behavior for overflow.
|
||||||
|
- **Rarity in many workloads.** Empirical studies (e.g., Hennessy & Patterson, *Computer Architecture: A Quantitative Approach*) report that integer divide and remainder instructions are uncommon (typically <1% of dynamic instructions), while multiplies are more frequent in HPC and crypto workloads.
|
||||||
|
|
||||||
|
For XH-1, the unit sits on the execution path of every core. With 128 cores, area and per-core energy dominate; raw single-thread latency of division matters less than aggregate throughput, die-area cost, and ease of verification.
|
||||||
|
|
||||||
|
## Existing Approaches
|
||||||
|
|
||||||
|
### Multiplication
|
||||||
|
|
||||||
|
1. **Iterative shift-and-add multiplier**
|
||||||
|
- One 64-bit adder reused across 64 cycles.
|
||||||
|
- Smallest area, lowest energy per multiplication, but very long latency.
|
||||||
|
|
||||||
|
2. **Array multiplier (combinational)**
|
||||||
|
- 64×64 array of full adders producing a 128-bit result.
|
||||||
|
- Single-cycle result, large area, high fan-out, and a long critical path.
|
||||||
|
- Historically too slow for one cycle at high clock frequencies.
|
||||||
|
|
||||||
|
3. **Wallace / Dadda tree**
|
||||||
|
- Tree of carry-save adders reducing partial products to two 128-bit vectors, then a final carry-propagate adder.
|
||||||
|
- Logarithmic depth; commonly used in high-performance cores.
|
||||||
|
- Larger area than array, but shorter critical path.
|
||||||
|
|
||||||
|
4. **Booth-encoded Wallace / Dadda**
|
||||||
|
- Radix-4 or higher Booth recoding reduces the number of partial products by ~2×.
|
||||||
|
- Common in modern cores (e.g., reported in implementations of ARM and x86 multipliers).
|
||||||
|
|
||||||
|
5. **Pipelined iterative multiplier**
|
||||||
|
- Splits the 64-cycle iterative multiplier into pipeline stages (commonly 2–4 stages).
|
||||||
|
- Used in many in-order RISC-V cores (e.g., the Rocket Chip generator's `MulDiv`).
|
||||||
|
|
||||||
|
6. **Dedicated single-cycle fused multiply-add (FMA) for integers**
|
||||||
|
- Rare for integer-only M-extensions; usually belongs to the F/D extensions.
|
||||||
|
|
||||||
|
### Division
|
||||||
|
|
||||||
|
1. **Restoring division**
|
||||||
|
- Classical shift-subtract. One bit per cycle, 64 cycles for 64-bit operands.
|
||||||
|
- Simple, easy to verify.
|
||||||
|
|
||||||
|
2. **Non-restoring division**
|
||||||
|
- Similar latency, but allows a single add/subtract per bit without explicit restore.
|
||||||
|
- Used in many textbook implementations.
|
||||||
|
|
||||||
|
3. **SRT division**
|
||||||
|
- Radix-4 or higher; produces 2+ bits per cycle using a small redundant quotient-digit table.
|
||||||
|
- Significantly faster (16–32 cycles for 64-bit) at higher area cost.
|
||||||
|
- Common in high-performance OoO cores (e.g., POWER, Itanium, recent x86).
|
||||||
|
|
||||||
|
4. **Newton–Raphson reciprocal + multiply**
|
||||||
|
- Iteratively refines an approximation of 1/d, then multiplies.
|
||||||
|
- Very high throughput once the reciprocal is available.
|
||||||
|
- Worst-case latency is higher than SRT; best for repeated divisions by the same divisor.
|
||||||
|
- Rare in integer pipelines due to initial latency.
|
||||||
|
|
||||||
|
5. **Goldschmidt division**
|
||||||
|
- Similar to Newton–Raphson; iteratively scales numerator and denominator toward 1.
|
||||||
|
- Same usage profile as Newton–Raphson.
|
||||||
|
|
||||||
|
6. **Lookup-table based constant dividers**
|
||||||
|
- For known small constant divisors, the compiler/runtime can replace DIV with a multiply-by-reciprocal.
|
||||||
|
- Microarchitectural implication: the DIV unit need not be heavily optimized if software frequently replaces division by constants.
|
||||||
|
|
||||||
|
## Alternative Designs
|
||||||
|
|
||||||
|
For the XH-1 MUL/DIV unit, four credible microarchitectural templates are considered.
|
||||||
|
|
||||||
|
### Design A: Shared iterative multi-cycle unit (Rocket-style)
|
||||||
|
|
||||||
|
- One 64-bit datapath reused for both MUL and DIV.
|
||||||
|
- MUL: 1 cycle/partial product (radix-2 Booth optional).
|
||||||
|
- DIV: 1 bit/cycle, restoring or non-restoring.
|
||||||
|
- MUL latency: ~33–35 cycles (radix-2 Booth) or ~64 cycles (plain shift-add).
|
||||||
|
- DIV latency: ~64 cycles.
|
||||||
|
- Throughput: 1 MUL or DIV per ~32 cycles (shared); MUL and DIV cannot execute concurrently.
|
||||||
|
- Plausible canonical reference: the `MulDiv` module in the BOOM/Rocket Chip generator (UC Berkeley).
|
||||||
|
|
||||||
|
### Design B: Pipelined iterative multiplier + iterative divider
|
||||||
|
|
||||||
|
- Multiplier: 2–4 stage pipelined radix-2/radix-4 iterative unit.
|
||||||
|
- Divider: separate 64-bit iterative datapath (non-restoring or SRT-radix-2).
|
||||||
|
- MUL throughput: 1 per cycle once pipeline is full.
|
||||||
|
- DIV throughput: 1 per ~32 cycles.
|
||||||
|
- Independent issue of MUL and DIV is possible.
|
||||||
|
|
||||||
|
### Design C: Pipelined Wallace/Booth multiplier + SRT-radix-4 divider
|
||||||
|
|
||||||
|
- Multiplier: 3-stage pipelined radix-4 Booth → Wallace tree → CPA. Produces full 128-bit result.
|
||||||
|
- Divider: radix-4 SRT producing 2 bits/cycle; ~17 cycles for 64-bit DIV.
|
||||||
|
- DIV/REM can be produced simultaneously since quotient digits are known.
|
||||||
|
- Area: significantly larger than Designs A and B.
|
||||||
|
- Latency: MUL ~3–4 cycles, DIV ~17–20 cycles.
|
||||||
|
- Used in many modern superscalar cores.
|
||||||
|
|
||||||
|
### Design F: Fused integer MAC / FMA
|
||||||
|
|
||||||
|
- Add an integer fused multiply-add returning low 64 bits (a*b)+c in one operation.
|
||||||
|
- This is non-standard for RV64M and would require custom opcodes or being staged behind a regular MUL+ADD sequence.
|
||||||
|
- Documented here for completeness, not recommended without strong workload evidence.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
| Property | A: Iterative shared | B: Pipelined iterative MUL + iterative DIV | C: Wallace/Booth MUL + SRT-4 DIV |
|
||||||
|
|---|---|---|---|
|
||||||
|
| MUL latency | ~33–64 cycles | 3–5 cycles | 3–4 cycles |
|
||||||
|
| MUL throughput | 1 / 32 cycles | 1 / cycle | 1 / cycle |
|
||||||
|
| DIV latency | ~64 cycles | ~32 cycles | ~16–20 cycles |
|
||||||
|
| DIV throughput | 1 / 64 cycles | 1 / 32 cycles | 1 / 16 cycles |
|
||||||
|
| 64×64→128 datapath | Yes (shared) | Yes (MUL only) | Yes (Wallace) |
|
||||||
|
| MUL+DIV concurrency | No (shared) | Yes (separate datapaths) | Yes (separate datapaths) |
|
||||||
|
| Estimated relative area | 1.0× | ~1.5–2.0× | ~3.0–5.0× |
|
||||||
|
| Estimated critical path | Short | Short | Longest (CPA final stage) |
|
||||||
|
| Verification complexity | Low | Medium | High |
|
||||||
|
| Fits "small in-order" model | Excellent | Good | Marginal |
|
||||||
|
|
||||||
|
ASSUMPTION: Area estimates above are rough order-of-magnitude relative numbers based on typical RISC-V implementations and the cited textbooks. They have not been measured for XH-1.
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
|
||||||
|
### Design A (iterative shared)
|
||||||
|
|
||||||
|
- Smallest area per core, which directly reduces die cost across 128 cores.
|
||||||
|
- Lowest per-core dynamic energy for the rare case of an actual MUL/DIV.
|
||||||
|
- Easiest to verify formally (small state space, one datapath).
|
||||||
|
- Matches the "many small cores" scaling philosophy.
|
||||||
|
- Canonical reference: Rocket Chip `MulDiv`.
|
||||||
|
|
||||||
|
### Design B (pipelined iterative MUL + iterative DIV)
|
||||||
|
|
||||||
|
- MUL throughput is high enough to support HPC and crypto workloads where 64-bit multiplies are common.
|
||||||
|
- DIV remains simple.
|
||||||
|
- Area increase over A is bounded.
|
||||||
|
- Two independent datapaths simplify scheduling in the issue stage.
|
||||||
|
|
||||||
|
### Design C (Wallace/Booth + SRT-4)
|
||||||
|
|
||||||
|
- Best raw latency and throughput for both operations.
|
||||||
|
- Suitable for single-core-bound workloads or for cores that need to hide memory latency behind fast arithmetic.
|
||||||
|
- DIV+REM can be produced together with little extra hardware.
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
|
||||||
|
### Design A
|
||||||
|
|
||||||
|
- DIV latency of 64 cycles is long; if the surrounding pipeline is short (e.g., 5–7 stages), the unit will dominate total execution time for any divide.
|
||||||
|
- Back-to-back MULs serialize.
|
||||||
|
- Under HPC or cryptography kernels, MUL throughput becomes a bottleneck.
|
||||||
|
|
||||||
|
### Design B
|
||||||
|
|
||||||
|
- More area than A.
|
||||||
|
- Pipelined MUL increases register pressure in the issue queue and requires more bypass paths in the surrounding execution stage.
|
||||||
|
- DIV still slow.
|
||||||
|
|
||||||
|
### Design C
|
||||||
|
|
||||||
|
- Largest area per core, replicated 128 times.
|
||||||
|
- Highest per-core power.
|
||||||
|
- Wallace tree and SRT have long critical paths that may limit clock frequency for the whole core.
|
||||||
|
- Verification complexity is significantly higher: partial-product reduction, Booth recoding, SRT quotient-digit selection tables, and divider corner cases (e.g., `INT64_MIN / -1`) all need separate coverage.
|
||||||
|
- Wall-clock design and verification cost may delay the whole project.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
|
||||||
|
PROPOSAL: For XH-1, an in-order core with 128 instances on die, the dominant design constraint is **per-core area, energy, and verification cost**, not single-thread peak performance. The MUL/DIV unit should therefore favor small, simple, well-trodden implementations.
|
||||||
|
|
||||||
|
Specific implications for XH-1:
|
||||||
|
|
||||||
|
- The 128-core factor means the MUL/DIV unit's area is multiplied by 128. Even a 2× area difference per core translates to a substantial absolute area delta.
|
||||||
|
- The energy of 128 MUL/DIV datapaths, even at low utilization, contributes to total socket power.
|
||||||
|
- A long-latency MUL/DIV unit is acceptable if the surrounding pipeline is deep enough or if it can be overlapped with other in-flight instructions in the same core.
|
||||||
|
- Single-cycle MUL would impose a critical path on the whole core; for a 128-core design, sustained high clock frequency across all cores is critical to total throughput.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
|
||||||
|
Scalability dimensions to consider:
|
||||||
|
|
||||||
|
- **Wiring and layout.** A 128-core die has a complex interconnect. A small MUL/DIV unit is easier to place and route within each core tile. Designs with large irregular adder trees (Wallace/SRT) complicate physical design at high core counts.
|
||||||
|
- **Verification replication.** Bugs in the MUL/DIV unit, if present, propagate to 128 cores. A simpler, formally verifiable design (Design A) is safer for replication.
|
||||||
|
- **Yield.** Smaller per-core area improves yield and binning flexibility; large per-core area reduces the number of cores that fit on a reticle at the target process node.
|
||||||
|
- **Power delivery.** 128 simultaneous MUL/DIV operations are unlikely, but worst-case power events (e.g., SIMD-style vector MUL workloads scaled down to integer MUL) must be within the socket's power-delivery budget.
|
||||||
|
- **Frequency scaling.** A 128-core chip with modest per-core frequency but high aggregate throughput may benefit from a short critical path. A Wallace multiplier's critical path can limit fmax for the whole core.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
|
||||||
|
- MUL/DIV instructions are infrequent in general-purpose workloads (often <1% dynamic instructions) but can dominate kernels in cryptography (AES, ChaCha20, RSA), big-integer arithmetic (GMP-style libraries), and some HPC kernels.
|
||||||
|
- If XH-1 is intended for general-purpose server or desktop use, the MUL/DIV unit will rarely be on the critical path of a thread.
|
||||||
|
- If XH-1 targets HPC or cryptography, the MUL throughput becomes important. In this case, B or C should be reconsidered.
|
||||||
|
- Software can use compiler transformations to replace DIV by constants with multiply-by-reciprocal, reducing pressure on the DIV unit.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
|
||||||
|
- **Sign handling.** The unit must correctly handle `MULH`, `MULHSU`, and `MULHU` as well as overflow cases of DIV (notably `INT64_MIN / -1`, which must produce `INT64_MIN` per RISC-V spec).
|
||||||
|
- **REM vs DIV.** Producing REM in parallel with DIV using the same datapath is standard in restoring/non-restoring designs; the unit should support issuing DIVU/REMU pairs in one operation.
|
||||||
|
- **Pipeline interface.** The unit must integrate with the core's issue, wakeup, and writeback stages. If in-order, the issue stage must stall in-order cores on multi-cycle MUL/DIV. If OoO, completion must wait for the unit's completion signal.
|
||||||
|
- **Bypassing.** Forwarding paths from the MUL/DIV pipeline registers to dependent instructions must be designed carefully to avoid structural hazards.
|
||||||
|
- **Early termination.** For DIV, the unit can terminate early when the remainder is zero, saving cycles. Implementation cost is low.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
|
||||||
|
- **Corner cases.** RV64M has well-defined corner cases: `INT64_MIN / -1`, division by zero, overflow in REM, sign interactions in MULH-family.
|
||||||
|
- **Directed + constrained-random.** A combination of directed tests for ISA corner cases and constrained-random for the rest is standard practice (e.g., as in the RISC-V architectural test framework, riscv-tests).
|
||||||
|
- **Formal verification.** A small iterative multiplier/divider (Design A) is amenable to formal proofs of correctness for a few-bit case and inductive scaling. A Wallace + SRT unit (Design C) is significantly harder to formally verify due to selector-table complexity.
|
||||||
|
- **Cross-core equivalence.** With 128 identical cores, regression in one core implies regression in all 128. A well-verified single-core design simplifies the chip-level verification effort.
|
||||||
|
- **Testbench reuse.** The RISC-V community maintains architectural compliance tests that should be run against the MUL/DIV unit regardless of the chosen design.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
|
||||||
|
PROPOSAL: Adopt a **Design B–leaning approach**: a small, simple, well-understood MUL/DIV unit similar in spirit to Rocket Chip's `MulDiv`, with the following characteristics:
|
||||||
|
|
||||||
|
- A **radix-4 Booth-encoded iterative multiplier** (or radix-2 if radix-4 proves too complex for the area budget) producing 64 bits of result per ~16 cycles, sharing partial datapath with the divider if needed.
|
||||||
|
- A **non-restoring (or radix-2 SRT) divider** completing in ~32 cycles.
|
||||||
|
- MUL and DIV on the same datapath with **shared state** but capable of being interleaved at issue time.
|
||||||
|
- Optional microarchitectural relaxation: a separate tiny **fast-MUL path for 32×32→64 results** (the low half of MUL where both operands are sign- or zero-extended from 32 bits) to accelerate common cases. This adds minimal area.
|
||||||
|
|
||||||
|
This recommendation is provisional and is the lightest-weight option that still keeps MUL throughput reasonable. It avoids the critical-path cost of Design C and the throughput limit of Design A, while remaining well within the verification budget of a 128-core project.
|
||||||
|
|
||||||
|
RECOMMENDATION: If workload analysis (not yet performed) shows MUL-heavy HPC/cryptography use, escalate to a pipelined radix-4 Booth multiplier with a 2–3 cycle latency, keeping the iterative divider. If workload analysis shows almost no MUL/DIV usage, drop to a plain Design A.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
|
||||||
|
- **Low–Medium** for any specific microarchitectural recommendation. The document is at an early stage; the recommendation will be revised after:
|
||||||
|
1. Workload analysis (target use cases of XH-1).
|
||||||
|
2. Synthesis of representative MUL/DIV units in the target technology.
|
||||||
|
3. Frequency, area, and power target constraints.
|
||||||
|
- **High** that the iterative, shared-datapath approach (Design A or B) is the appropriate starting point for a 128-core, area-constrained, verification-constrained design.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- What is the target frequency of XH-1 cores, and what is the critical-path budget for the MUL/DIV unit?
|
||||||
|
- What process node is targeted, and what is the per-core area budget?
|
||||||
|
- What is the intended workload mix (server, HPC, embedded, ML)?
|
||||||
|
- Is the core in-order or out-of-order? The MUL/DIV interface and latency tolerance depend strongly on this.
|
||||||
|
- Will the F extension (floating-point) be present in the same core, and if so, should integer MUL be reused inside an FMA datapath?
|
||||||
|
- Will the V extension (vector) be present? If so, scalar integer MUL may be lightly used and the scalar MUL/DIV unit can be minimal.
|
||||||
|
- Are fused integer MAC operations required by any target workload?
|
||||||
|
- What level of formal verification is mandated for XH-1?
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
Primary and authoritative references used or cited in this document:
|
||||||
|
|
||||||
|
- RISC-V International, *The RISC-V Instruction Set Manual, Volume I: Unprivileged Architecture* — official definition of RV64M (MUL, MULH, MULHSU, MULHU, DIV, DIVU, REM, REMU) and division overflow semantics.
|
||||||
|
- RISC-V International, *Architectural Compatibility Test Suite* (riscv-tests, riscv-arch-test) — official compliance test references.
|
||||||
|
- UC Berkeley Architecture Research, *Rocket Chip Generator* documentation — reference for the small iterative `MulDiv` module.
|
||||||
|
- UC Berkeley Architecture Research, *BOOM Out-of-Order Processor* documentation — reference for SRT-class dividers and pipelined multipliers in BOOM v2/v3.
|
||||||
|
- Hennessy & Patterson, *Computer Architecture: A Quantitative Approach* (recent editions) — workload frequency of MUL/DIV, energy/area considerations.
|
||||||
|
- Ercegovac & Lang, *Digital Arithmetic* — comprehensive treatment of shift-add, Booth, Wallace, SRT, and Newton–Raphson dividers.
|
||||||
|
- Parhami, *Computer Arithmetic: Algorithms and Hardware Designs* — additional reference for multiplier and divider architectures.
|
||||||
|
|
||||||
|
ASSUMPTION: Specific page numbers and edition identifiers for Hennessy & Patterson, Ercegovac & Lang, and Parhami have not been quoted above because the exact editions in the XH-1 research library have not been recorded in this document. They should be cited precisely when this document is finalized.
|
||||||
|
|
||||||
|
INSUFFICIENT EVIDENCE: No synthesis, layout, or PPA data for the target process node is yet available for any of the four design candidates. The relative area and energy figures are qualitative estimates only.
|
||||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,37 @@
|
|||||||
|
VERDICT: FAIL
|
||||||
|
|
||||||
|
ISSUES:
|
||||||
|
- **Design B–leaning recommendation contradicts the stated proposal and the comparison table.** The "XH-1 Considerations" and "128-Core Scalability" sections argue decisively that area, energy, verification, and 128× replication favor the smallest, simplest design (Design A). The recommendation then promotes Design B (or a Design B–leaning hybrid) without reconciling the conflict. The justification "avoids the critical-path cost of Design C and the throughput limit of Design A" is hand-waving; the comparison table shows Design A's critical path is shortest.
|
||||||
|
- **The four candidate designs are labeled inconsistently.** The "Existing Approaches" section enumerates multiplication and division schemes separately, but the "Alternative Designs" section labels them A, B, C, then jumps to "Design F" (fused MAC) without a Design D or E. This is sloppy and obscures the design space — e.g., where is the pipelined iterative multiplier with SRT-radix-4 divider option? Where is a non-pipelined Wallace+non-restoring hybrid?
|
||||||
|
- **Design B is mischaracterized in the comparison table.** "Pipelined iterative MUL + iterative DIV" with "MUL latency 3–5 cycles" and "1/cycle" throughput is not really an "iterative" multiplier — that description fits a pipelined combinational or partially combinational datapath. The terminology is inconsistent with the rest of the document and with how Rocket's `MulDiv` actually works (which is closer to Design A, not B).
|
||||||
|
- **Unsupported area/energy claims.** "Estimated relative area 1.0× / ~1.5–2.0× / ~3.0–5.0×" and the corresponding energy numbers are presented as quasi-data with no synthesis backing, no source citation, and no process node. The document acknowledges this in a footnote-style ASSUMPTION, but the numbers are still used as if comparable in the comparison table and in the recommendation. For a research document whose central trade-off is area-per-core × 128, this is a serious gap.
|
||||||
|
- **Missing alternative: constant-divider optimization (magic-number multiplication).** The "Existing Approaches" section mentions lookup-table / multiply-by-reciprocal constant dividers in passing, but no design candidate explores offloading DIV-by-constant entirely to software or a dedicated microarchitectural helper. Given that compilers routinely do this and the document explicitly states MUL/DIV are rare, omitting this as a real alternative is a gap.
|
||||||
|
- **Missing alternative: shared cluster-level MUL/DIV unit.** For a 128-core design, a recurring microarchitectural option is to share one (or a few) high-performance MUL/DIV units across a cluster of cores (e.g., 4 or 8 cores share a Wallace/SRT unit), trading per-core latency for die area. The document does not consider this at all, even though it is highly relevant to the stated scaling concern. This is a meaningful missing alternative.
|
||||||
|
- **Missing alternative: vector / SIMD reuse.** The document explicitly defers vector MUL/MAC but does not consider whether the scalar MUL/DIV unit should be designed knowing that the V extension (if present) would dominate aggregate multiply throughput. The "Open Questions" section flags this but it should be in the alternatives analysis.
|
||||||
|
- **Failure to consider 128 cores in the recommendation logic.** The recommendation says "B-leaning" while the 128-core analysis section argues for "A." The document never reconciles these; it just presents both without resolving. This is an internal contradiction.
|
||||||
|
- **The "fast 32×32→64 MUL path" recommendation is unsupported.** Stated as "minimal area" with no area estimate and no discussion of how it interacts with the shared datapath, the issue logic, the partial-product generator, or verification. It is a non-trivial addition that needs justification.
|
||||||
|
- **In-order vs out-of-order core assumption is missing.** The recommendation is presented as if the core is in-order, but the document never states the core microarchitecture. Most of the latency-vs-throughput trade-off and the Wallace/SRT critical-path analysis hinges on this. This is a critical missing assumption.
|
||||||
|
- **"Short critical path" claim for the iterative divider is unsupported.** A 64-bit non-restoring divider has a 64-bit adder in its critical path, which is comparable to other units in the core. The document asserts the iterative approach has the shortest critical path without discussing the actual datapath depth of the 64-bit CPA inside the divider loop.
|
||||||
|
- **"DIV/REM produced simultaneously" claim for SRT-radix-4 is misleading.** SRT produces quotient digits; the remainder is recovered at the end from a redundant form and typically requires a correction step. Stating "DIV/REM can be produced simultaneously since quotient digits are known" elides this and is borderline incorrect.
|
||||||
|
- **Wallace tree is described as having a "longer critical path" than array, which is backwards.** A Wallace tree has a *logarithmic* depth and a strictly shorter critical path than the linear array; the trade-off is area/routing, not critical path. This is a factual error.
|
||||||
|
- **"Radix-4 Booth reduces partial products by ~2×" is sloppy.** Radix-4 Booth recodes 64 bits into 33 signed digits, reducing partial products from 64 to 33 — roughly 2× fewer, but the document should state the exact number. Minor, but symptomatic.
|
||||||
|
- **Citation hygiene is poor.** "Hennessy & Patterson" and "Ercegovac & Lang" are cited with no specific edition, chapter, page, or even which textbook (H&P has multiple versions; "recent editions" is not a citation). For a research document, this is a serious sourcing weakness, and the document itself acknowledges the gap.
|
||||||
|
- **"Reported in implementations of ARM and x86 multipliers"** — no specific ARM core or x86 microarchitecture is named. This is a hallucination-prone claim with no source.
|
||||||
|
- **"Empirical studies (e.g., Hennessy & Patterson) report that integer divide and remainder instructions are uncommon (typically <1%)"** — the specific number and study are not cited. Different workloads vary widely; some SPEC int workloads have higher divide frequency. Unsupported.
|
||||||
|
- **Workload analysis is repeatedly deferred but recommendations are made anyway.** The recommendation, the proposal, and the open questions all say "workload analysis pending" yet a specific microarchitectural recommendation is still made. This is unjustified given the document's own caveats.
|
||||||
|
- **No quantitative power analysis at all.** For a 128-core design, dynamic and leakage power of replicated MUL/DIV units should be at least estimated, even at a high level. The document says "energy" repeatedly but provides no numbers.
|
||||||
|
- **The 128-core verification claim ("formal proofs of correctness for a few-bit case and inductive scaling") is a hand-wave.** Inductive scaling proofs of arbitrary Booth/SRT/non-restoring dividers are not standard practice and are not trivial. This overstates the state of formal verification for these units.
|
||||||
|
|
||||||
|
REQUIRED FIXES:
|
||||||
|
- Resolve the contradiction between the "XH-1 Considerations" argument (which favors Design A) and the recommendation (which favors Design B). Either justify the deviation explicitly or align the recommendation with the analysis.
|
||||||
|
- Fix the factual error about Wallace trees (shorter, not longer, critical path than array multipliers) and the misleading SRT remainder-claim.
|
||||||
|
- Renumber design candidates consistently (A, B, C, D, …) and include the missing alternatives: cluster-shared MUL/DIV, constant-divider / magic-number offload, and a hybrid Wallace + non-restoring option.
|
||||||
|
- Provide actual synthesis, area, and energy estimates from at least one reference technology (e.g., a published RISC-V core, or a synthesized estimate from open-source IP like Rocket/BOOM) rather than qualitative multipliers with no anchor.
|
||||||
|
- Add a discussion of how the recommendation depends on in-order vs out-of-order core microarchitecture, and explicitly state the assumed core type for this document.
|
||||||
|
- Provide specific citations (edition, chapter/page) for Hennessy & Patterson, Ercegovac & Lang, and Parhami, or remove the claims that depend on them.
|
||||||
|
- Either substantiate or remove the "32×32 fast MUL" recommendation; it is currently an unjustified adder.
|
||||||
|
- Include a power/energy analysis (even rough) for 128-core replication; this is central to the stated design constraint.
|
||||||
|
- Quantify or remove the "1% dynamic instructions" claim; cite the specific study and workload.
|
||||||
|
- Address verification claims more carefully: do not overstate the practicality of inductive formal proofs for divider corner cases, and note the realistic state of formal vs constrained-random coverage for each design.
|
||||||
|
|
||||||
|
CONFIDENCE: HIGH
|
||||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,2 @@
|
|||||||
|
2026-08-25T18:09:56Z research/03-core-design/mul-div-unit.md 1 research completed
|
||||||
|
2026-08-25T18:10:12Z research/03-core-design/mul-div-unit.md 1 review api-failure
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
research/03-core-design/mul-div-unit.md
|
||||||
+194
@@ -0,0 +1,194 @@
|
|||||||
|
# Multiplier–Divider Unit for the XH-1 Core
|
||||||
|
|
||||||
|
## Status
|
||||||
|
DRAFT — Engineering research, no silicon data.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
This document investigates the design of an integer multiply/divide (MUL/DIV) unit suitable for one tile of the XH-1 128-core RISC-V processor. It frames the problem space (RV32M/RV64M, latency vs. throughput, area and energy), surveys the canonical microarchitectural options (iterative shift-and-add, array multipliers, Booth/Wallace/Dadda trees, reciprocal/sRT dividers, radix-2/4/8 dividers, and merged MAC-fused units), and maps them against the constraints implied by the other XH-1 core documents (small per-core budget, 128-core replication, expected integration with the pipeline, register file, and load–store unit). The document is written so that downstream choices about pipeline depth, core clock target, and shared vs. private FP/MAC can be made coherently with the MUL/DIV decision.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
What microarchitecture for the integer multiply/divide unit best fits one XH-1 core, given that the core will be replicated 128 times, must support the M extension on RV32 and/or RV64, and must coexist with a pipeline, register file, and load–store unit whose budgets are not yet fixed?
|
||||||
|
|
||||||
|
Sub-questions:
|
||||||
|
1. Should the unit be fully combinatorial, multi-cycle iterative, or pipelined?
|
||||||
|
2. Which radix and which encoding (Booth-2, Booth-3, modified Booth-4) is appropriate for the expected operand width?
|
||||||
|
3. Should MUL and DIV share silicon, or be separate datapaths?
|
||||||
|
4. How does the choice interact with 128-core replication (area amortization, frequency, yield)?
|
||||||
|
5. What are the verification implications of each choice?
|
||||||
|
|
||||||
|
## Background
|
||||||
|
The RISC-V M extension defines four signed/unsigned multiply variants producing 2X-bit results (MUL, MULH, MULHU, MULHSU), four matching multiply-high variants, and signed/unsigned divide and remainder (DIV, DIVU, REM, REMU). On RV64 there is an additional MULW/DIVW/REM family that operates on 32-bit values and sign-extends.
|
||||||
|
|
||||||
|
Key microarchitectural properties that drive the design:
|
||||||
|
- Latency-tolerance: in a deep pipeline, multi-cycle iterative units are acceptable as long as the structural hazard is bounded and the scoreboard/issue logic handles in-flight multiplies.
|
||||||
|
- Throughput: the M extension opcodes are infrequent in many workloads but dominate in linear-algebra kernels, crypto, and hash functions.
|
||||||
|
- Operand width: an RV64 core needs both 64-bit and 32-bit paths. A 64×64→128 multiplier is roughly 4× the area of a 32×32→64 multiplier.
|
||||||
|
- Result width: MULH-family instructions require the full 2X-bit result; MUL only requires the low X bits. Most designs share the upper datapath and select the low half.
|
||||||
|
|
||||||
|
ASSUMPTION: XH-1 cores implement the M extension. If only the I extension is required, the MUL/DIV unit collapses dramatically. The remainder of this document assumes M is in scope.
|
||||||
|
ASSUMPTION: XH-1 is RV64. If RV32 only, all 64-bit-specific considerations below can be relaxed.
|
||||||
|
|
||||||
|
## Existing Approaches
|
||||||
|
|
||||||
|
### Multipliers
|
||||||
|
1. Iterative shift-and-add multiplier
|
||||||
|
2. Array (braid) multiplier
|
||||||
|
3. Wallace tree multiplier
|
||||||
|
4. Dadda tree multiplier
|
||||||
|
5. Booth-encoded tree multiplier (radix-4, modified Booth)
|
||||||
|
6. Higher-radix Booth (radix-8/16) with 4:2/5:2 compressor trees
|
||||||
|
7. Pipelined versions of any of the above, with stage counts from 1 to N
|
||||||
|
|
||||||
|
### Dividers
|
||||||
|
1. Restoring divider
|
||||||
|
2. Non-restoring divider
|
||||||
|
3. SRT divider (s radix-2, radix-4, radix-8, radix-16)
|
||||||
|
4. Newton–Raphson reciprocal + multiply (software or hardware)
|
||||||
|
5. Goldschmidt divider
|
||||||
|
6. Digit-recurrence with prescaling (for faster convergence)
|
||||||
|
7. Lookup-table (LUT) assisted dividers (small LUT, big LUT)
|
||||||
|
|
||||||
|
PROPOSAL: Classify each option along four axes — latency (cycles), throughput (1/n per cycle), area (gate equivalent or µm² estimate), and design complexity (verification effort, corner cases).
|
||||||
|
|
||||||
|
## Alternative Designs
|
||||||
|
|
||||||
|
### Option A: Pipelined iterative multiplier + iterative non-restoring divider
|
||||||
|
- One shared 64-bit datapath, Booth-2 radix-4, latency ≈ 3–4 cycles for MUL, fully pipelined at 1 result/cycle.
|
||||||
|
- Divider: non-restoring, ≈ 32–64 cycles for RV64 DIV, variable.
|
||||||
|
- Area: smallest among the options; one of the smallest in published small-core implementations.
|
||||||
|
- Verification: standard; iterative state machines are well understood.
|
||||||
|
|
||||||
|
### Option B: Fully combinational 64×64 array multiplier + SRT-4 divider
|
||||||
|
- MUL latency: 1 cycle, but very high area and long critical path.
|
||||||
|
- DIV: SRT-4, ≈ 16 cycles typical.
|
||||||
|
- Area: largest.
|
||||||
|
- Verification: simple timing closure problem, but 128 cores × this area is likely prohibitive.
|
||||||
|
|
||||||
|
### Option C: Pipelined Wallace/Booth tree (3-stage) + SRT-4 divider
|
||||||
|
- MUL: 3-cycle pipelined, 1 result/cycle sustained.
|
||||||
|
- DIV: 16–20 cycles, fully pipelined divider.
|
||||||
|
- Area: moderate to large; tree is irregular.
|
||||||
|
- Verification: irregular partial-product reduction is harder to verify than array.
|
||||||
|
|
||||||
|
### Option D: Fused multiply–accumulate (MAC) with Booth-3 (radix-8) and 4:2 compressors
|
||||||
|
- MUL: 3-cycle, also accepts an accumulate operand each cycle.
|
||||||
|
- Provides a useful primitive for dot-product kernels and soft-fp libraries.
|
||||||
|
- Divider: separate iterative non-restoring path.
|
||||||
|
- Area: similar to C, but the accumulator latch and bypass network add cost.
|
||||||
|
|
||||||
|
### Option E: Truncated 32×32→64 only, with MULHW/DIVW synthesized in microcode
|
||||||
|
- Illegal in general: software synthesis of MULH is too slow to be acceptable.
|
||||||
|
- Listed for completeness and to discard.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
| Option | MUL latency | MUL throughput | DIV latency (RV64) | Area (relative) | Design complexity |
|
||||||
|
|--------|-------------|----------------|---------------------|------------------|--------------------|
|
||||||
|
| A — Iterative Booth + NR div | 3–4 cyc | 1/cyc pipelined | 32–64 cyc | 0.6–0.8× | Low |
|
||||||
|
| B — Array + SRT-4 | 1 cyc | 1/cyc | 16 cyc | 2.5–3.0× | Medium (timing) |
|
||||||
|
| C — Wallace/Booth + SRT-4 | 3 cyc | 1/cyc | 16–20 cyc | 1.0–1.2× | High (tree) |
|
||||||
|
| D — MAC (Booth-3, 4:2) + NR div | 3 cyc | 1/cyc (with acc) | 32–64 cyc | 1.1–1.3× | High |
|
||||||
|
| E — Truncated, microcoded | n/a | n/a | n/a | smallest | n/a (incomplete) |
|
||||||
|
|
||||||
|
Numbers above are ORDER-OF-MAGNITUDE ESTIMATES based on published small-core implementations; INSUFFICIENT EVIDENCE exists to claim a specific gate count or µm² without synthesis at a known target node.
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
- Option A: smallest area per core → most 128-core replication headroom; trivially fits any pipeline depth; well-understood verification.
|
||||||
|
- Option B: best single-cycle latency; useful for OoO cores with tight issue windows.
|
||||||
|
- Option C: good balance of latency and area; widely used in commercial OoO cores.
|
||||||
|
- Option D: enables efficient dot-product and SIMD-style software; helps a soft FP stack.
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
- Option A: lower peak MUL throughput than pipelined trees; multi-cycle DIV may stall the issue queue on back-to-back DIVs.
|
||||||
|
- Option B: area is prohibitive at 128 cores; long critical path forces a slow core clock or deep pipelining (defeating the latency benefit).
|
||||||
|
- Option C: Wallace/Dadda partial-product reduction has irregular carry-save structures that are harder to formally verify and harder to fix in ECO.
|
||||||
|
- Option D: accumulator adds bypass and forwarding complexity into the pipeline and register-file writeback path; benefits only workloads that can be rewritten to use the MAC.
|
||||||
|
- Option E: not viable for an M-extension-compliant core.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
ASSUMPTION: XH-1 is a tiled, replicated design where per-core area is a first-class constraint because the whole 128-core array must fit in the package, power, and yield envelope. A small per-core MUL/DIV is therefore a high-value design point.
|
||||||
|
|
||||||
|
PROPOSAL: Treat MUL/DIV as one of the "shared-tile resource" candidates. Specifically:
|
||||||
|
- If 128 cores × 1 MUL/DIV per core exceeds the area budget, fall back to a per-cluster shared unit (e.g., 1 MUL/DIV per 4 or 8 cores) behind a dedicated interconnect port.
|
||||||
|
- The crossbar/coherence documents in the repository should be checked before committing to a per-core vs. shared decision.
|
||||||
|
- A shared unit complicates the scoreboard: the issue queue must track remote MUL/DIV latency, which can be 4–8× the per-core MUL latency.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
- Per-core area: Option A scales best; Option B is the worst case.
|
||||||
|
- Frequency: Option A and Option C both close timing at typical small-core targets; Option B is the only option likely to force a slower core clock.
|
||||||
|
- Yield: small per-core datapath → higher core yield, fewer fatal defects per die.
|
||||||
|
- Coherence traffic: a long-latency MUL/DIV keeps the core stalled but does not generate coherence traffic; an off-core MUL/DIV (shared) does, because the issuing core may continue past the result and the unit must return through the interconnect.
|
||||||
|
- Verification: per-core unit × 128 = 128 instances × one verification suite. This is a strong argument for the simplest microarchitecture that meets the latency target.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
- For most non-numeric workloads, MUL/DIV throughput does not bound IPC; latency does not either, because the operand is rarely on the critical path. A 3–4 cycle MUL and a 32–64 cycle DIV is acceptable.
|
||||||
|
- For numeric, crypto, and hash kernels, MUL throughput is critical; the pipelined-tree options (C, D) win by roughly 2–3× peak throughput.
|
||||||
|
- DIV is rarely on the hot path; a 32–64 cycle iterative divider is almost always sufficient.
|
||||||
|
|
||||||
|
RECOMMENDATION (conditional): If XH-1 targets general-purpose + occasional numeric, prefer Option A. If XH-1 targets numeric/CV/crypto explicitly, prefer Option C.
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
- A 64×64→128 array multiplier at a modern node is roughly the size of the register file's writeback port plus the ALU; this is the dominant per-core cost in Option B.
|
||||||
|
- A pipelined 3-stage tree (Option C) shrinks the per-stage critical path at the cost of three sets of partial-product reduction and accumulation latches.
|
||||||
|
- An iterative multiplier (Option A) is dominated by a single 33–64-bit adder and a shift register, and is the smallest.
|
||||||
|
- PROPOSAL: Capture the per-option area in the implementation document once the target node is fixed. Until then, treat the "relative" column in the comparison table as the working estimate.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
- Combinational array multipliers (Option B) toggle the entire partial-product array every cycle; energy per MUL is highest.
|
||||||
|
- Pipelined tree multipliers (Option C) distribute the switching across pipeline registers, lowering per-cycle peak power but with similar energy per operation.
|
||||||
|
- Iterative shift-and-add (Option A) is the lowest per-operation energy because the active datapath per cycle is small (one adder stage).
|
||||||
|
- 128-core replication: peak power is the product of per-core dynamic power and number of active cores. Option A keeps per-core power lowest, which is the most important axis for a 128-core envelope.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
- Sign handling: MULHSU requires signed×unsigned with full sign extension. The most common bug source in MUL/DIV is signed/unsigned mode selection; design the control path so that mode bits are sourced from the decoder, never from a sticky register.
|
||||||
|
- Zero-detection: DIV by zero must complete in bounded time and write a defined result. The IEEE / RISC-V rule is that DIV/REM by zero return −1 / the dividend. The unit must NOT trap on divide-by-zero; that is a software choice.
|
||||||
|
- Overflow: for DIV, the only overflow case is INT_MIN / −1. RISC-V mandates that this returns INT_MIN. The divider must detect this explicitly; iterative non-restoring dividers do, but SRT designs must be checked.
|
||||||
|
- Latency variability: DIV latency is data-dependent in non-restoring designs (it is fixed in SRT). If the pipeline assumes a fixed MUL/DIV latency, prefer an SRT divider or a constant-iteration iterative divider.
|
||||||
|
- Reset and scan: a 128-core replication multiplies the scan chain length. PROPOSAL: gate scan on the MUL/DIV unit to limit shift power, at the cost of reduced fault coverage. This is a verification trade-off.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
- The MUL/DIV unit has the highest ratio of corner cases to lines of RTL of any execution unit. Famous verification pitfalls:
|
||||||
|
- Signed multiplication: 0 × INT_MIN, INT_MIN × INT_MIN.
|
||||||
|
- MULH overflow into the high half with sign extension.
|
||||||
|
- MULHSU sign/unsigned mixing.
|
||||||
|
- DIV by zero, REM by zero.
|
||||||
|
- DIV overflow (INT_MIN / −1, INT_MIN % −1).
|
||||||
|
- All 16 combinations of signed/unsigned × 4 ops × 2 widths on RV64.
|
||||||
|
- PROPOSAL: Maintain a directed-test corpus at the unit level that exhausts the (sign × op × corner-operand) matrix, and a constrained-random suite at the core level.
|
||||||
|
- ASSUMPTION: XH-1 follows the riscv-formal convention of writing the MUL/DIV shadow model in a functional language. If so, the shadow model is the most expensive deliverable in the verification flow and is the strongest argument for picking the simplest microarchitecture.
|
||||||
|
|
||||||
|
## Software Considerations
|
||||||
|
- The compiler's latency model for MUL/DIV must match the hardware's; otherwise the scheduler will insert unnecessary stalls or miss scheduling opportunities. If Option A is chosen, the compiler should treat MUL as a 3–4 cycle latency op and DIV as a 32–64 cycle latency op.
|
||||||
|
- If MUL/DIV is moved off-core (shared), the toolchain needs an intrinsic or scheduling model that reflects the round-trip latency through the interconnect. This is non-trivial; most toolchains do not model non-uniform functional unit latency across cores.
|
||||||
|
- The Linux kernel's alternatives patching and the C library's soft-float paths sometimes use MUL/DIV in hot paths; verify that the chosen unit's latency is acceptable for the kernel configurations that the project intends to boot.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
PROPOSAL: For XH-1's first-pass implementation, adopt **Option A** (iterative Booth-2 multiplier pipelined to 1 result/cycle, non-restoring divider at 1 bit/cycle) for the following reasons:
|
||||||
|
1. Per-core area is minimized, which is the dominant axis in a 128-core replication.
|
||||||
|
2. The verification footprint is the smallest among the viable options.
|
||||||
|
3. The latency/throughput profile is acceptable for the majority of non-numeric workloads.
|
||||||
|
4. The microarchitecture is the easiest to ECO and to port across nodes.
|
||||||
|
|
||||||
|
PROPOSAL: If benchmarks (TBD) show that MUL/DIV throughput is on the critical path, migrate to **Option C** (pipelined 3-stage Booth/Wallace tree + SRT-4 divider) as a follow-on revision. Reserve **Option B** (fully combinational) for a hypothetical single-core "XH-1 Big" variant where replication is not the binding constraint.
|
||||||
|
|
||||||
|
ASSUMPTION: This recommendation is contingent on the pipeline depth, target clock, and per-core area budget not being fixed by other documents in the repository. If, for example, the pipeline document commits to a 5+ GHz target, Option A may not close timing and Option C becomes the floor rather than the upgrade.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
- Low-to-medium on the relative area numbers; they are order-of-magnitude estimates from published small-core work, not XH-1 synthesis.
|
||||||
|
- High on the qualitative trade-offs (area, verification complexity, replication cost).
|
||||||
|
- Low on the latency numbers, which depend on the target node and the adder architecture that is not yet specified in `alu.md` or `datapath.md`.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
- What is the target node and the target core clock? This gates the latency-vs-area decision.
|
||||||
|
- Is XH-1 RV32 or RV64? If RV32, the 32×32→64 multiplier is dramatically smaller.
|
||||||
|
- Is the M extension mandatory for the first tape-out, or can it be gated behind a config bit?
|
||||||
|
- Is there a plan for a shared per-cluster MAC/vector unit that subsumes the MUL/DIV? If so, the per-core MUL/DIV may be reduced to a minimal subset.
|
||||||
|
- What is the pipeline depth (see `pipeline.md`, currently SOON)? This determines whether a 1-cycle, 3-cycle, or N-cycle MUL fits cleanly into the execute stage.
|
||||||
|
- What is the issue model — in-order (single-issue, dual-issue, etc.)? The structural hazard model for MUL/DIV depends on it.
|
||||||
|
- Does the project plan to use `riscv-formal` for verification, or a UVM/constrained-random flow? The two flows have different optimal unit-level architectures.
|
||||||
|
- What are the integer workload targets — server, embedded, HPC, ML inference? Each implies a different MUL/DIV throughput requirement.
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
- The RISC-V "Unprivileged ISA" specification, M-extension chapter — INSUFFICIENT EVIDENCE to cite a specific revision without checking the repository. The relevant normative requirements are: MUL/MULH/MULHU/MULHSU produce full 2X-bit results, DIV/DIVU/REM/REMU round toward zero, DIV/REM by zero write back defined results, and INT_MIN / −1 is defined.
|
||||||
|
- Canonical microarchitecture textbooks (Hennessy & Patterson; Parhami; Ercegovac & Lang) — referenced only at the level of well-known technique names (Booth, Wallace, Dadda, SRT, non-restoring division). No specific page or claim is attributed.
|
||||||
|
- INSUFFICIENT EVIDENCE: any XH-1-specific synthesis, layout, or benchmark data. The comparison table is built from public-domain technique properties and must be re-validated with synthesis at the target node before tape-out.
|
||||||
+194
@@ -0,0 +1,194 @@
|
|||||||
|
# Multiplier–Divider Unit for the XH-1 Core
|
||||||
|
|
||||||
|
## Status
|
||||||
|
DRAFT — Engineering research, no silicon data.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
This document investigates the design of an integer multiply/divide (MUL/DIV) unit suitable for one tile of the XH-1 128-core RISC-V processor. It frames the problem space (RV32M/RV64M, latency vs. throughput, area and energy), surveys the canonical microarchitectural options (iterative shift-and-add, array multipliers, Booth/Wallace/Dadda trees, reciprocal/sRT dividers, radix-2/4/8 dividers, and merged MAC-fused units), and maps them against the constraints implied by the other XH-1 core documents (small per-core budget, 128-core replication, expected integration with the pipeline, register file, and load–store unit). The document is written so that downstream choices about pipeline depth, core clock target, and shared vs. private FP/MAC can be made coherently with the MUL/DIV decision.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
What microarchitecture for the integer multiply/divide unit best fits one XH-1 core, given that the core will be replicated 128 times, must support the M extension on RV32 and/or RV64, and must coexist with a pipeline, register file, and load–store unit whose budgets are not yet fixed?
|
||||||
|
|
||||||
|
Sub-questions:
|
||||||
|
1. Should the unit be fully combinatorial, multi-cycle iterative, or pipelined?
|
||||||
|
2. Which radix and which encoding (Booth-2, Booth-3, modified Booth-4) is appropriate for the expected operand width?
|
||||||
|
3. Should MUL and DIV share silicon, or be separate datapaths?
|
||||||
|
4. How does the choice interact with 128-core replication (area amortization, frequency, yield)?
|
||||||
|
5. What are the verification implications of each choice?
|
||||||
|
|
||||||
|
## Background
|
||||||
|
The RISC-V M extension defines four signed/unsigned multiply variants producing 2X-bit results (MUL, MULH, MULHU, MULHSU), four matching multiply-high variants, and signed/unsigned divide and remainder (DIV, DIVU, REM, REMU). On RV64 there is an additional MULW/DIVW/REM family that operates on 32-bit values and sign-extends.
|
||||||
|
|
||||||
|
Key microarchitectural properties that drive the design:
|
||||||
|
- Latency-tolerance: in a deep pipeline, multi-cycle iterative units are acceptable as long as the structural hazard is bounded and the scoreboard/issue logic handles in-flight multiplies.
|
||||||
|
- Throughput: the M extension opcodes are infrequent in many workloads but dominate in linear-algebra kernels, crypto, and hash functions.
|
||||||
|
- Operand width: an RV64 core needs both 64-bit and 32-bit paths. A 64×64→128 multiplier is roughly 4× the area of a 32×32→64 multiplier.
|
||||||
|
- Result width: MULH-family instructions require the full 2X-bit result; MUL only requires the low X bits. Most designs share the upper datapath and select the low half.
|
||||||
|
|
||||||
|
ASSUMPTION: XH-1 cores implement the M extension. If only the I extension is required, the MUL/DIV unit collapses dramatically. The remainder of this document assumes M is in scope.
|
||||||
|
ASSUMPTION: XH-1 is RV64. If RV32 only, all 64-bit-specific considerations below can be relaxed.
|
||||||
|
|
||||||
|
## Existing Approaches
|
||||||
|
|
||||||
|
### Multipliers
|
||||||
|
1. Iterative shift-and-add multiplier
|
||||||
|
2. Array (braid) multiplier
|
||||||
|
3. Wallace tree multiplier
|
||||||
|
4. Dadda tree multiplier
|
||||||
|
5. Booth-encoded tree multiplier (radix-4, modified Booth)
|
||||||
|
6. Higher-radix Booth (radix-8/16) with 4:2/5:2 compressor trees
|
||||||
|
7. Pipelined versions of any of the above, with stage counts from 1 to N
|
||||||
|
|
||||||
|
### Dividers
|
||||||
|
1. Restoring divider
|
||||||
|
2. Non-restoring divider
|
||||||
|
3. SRT divider (s radix-2, radix-4, radix-8, radix-16)
|
||||||
|
4. Newton–Raphson reciprocal + multiply (software or hardware)
|
||||||
|
5. Goldschmidt divider
|
||||||
|
6. Digit-recurrence with prescaling (for faster convergence)
|
||||||
|
7. Lookup-table (LUT) assisted dividers (small LUT, big LUT)
|
||||||
|
|
||||||
|
PROPOSAL: Classify each option along four axes — latency (cycles), throughput (1/n per cycle), area (gate equivalent or µm² estimate), and design complexity (verification effort, corner cases).
|
||||||
|
|
||||||
|
## Alternative Designs
|
||||||
|
|
||||||
|
### Option A: Pipelined iterative multiplier + iterative non-restoring divider
|
||||||
|
- One shared 64-bit datapath, Booth-2 radix-4, latency ≈ 3–4 cycles for MUL, fully pipelined at 1 result/cycle.
|
||||||
|
- Divider: non-restoring, ≈ 32–64 cycles for RV64 DIV, variable.
|
||||||
|
- Area: smallest among the options; one of the smallest in published small-core implementations.
|
||||||
|
- Verification: standard; iterative state machines are well understood.
|
||||||
|
|
||||||
|
### Option B: Fully combinational 64×64 array multiplier + SRT-4 divider
|
||||||
|
- MUL latency: 1 cycle, but very high area and long critical path.
|
||||||
|
- DIV: SRT-4, ≈ 16 cycles typical.
|
||||||
|
- Area: largest.
|
||||||
|
- Verification: simple timing closure problem, but 128 cores × this area is likely prohibitive.
|
||||||
|
|
||||||
|
### Option C: Pipelined Wallace/Booth tree (3-stage) + SRT-4 divider
|
||||||
|
- MUL: 3-cycle pipelined, 1 result/cycle sustained.
|
||||||
|
- DIV: 16–20 cycles, fully pipelined divider.
|
||||||
|
- Area: moderate to large; tree is irregular.
|
||||||
|
- Verification: irregular partial-product reduction is harder to verify than array.
|
||||||
|
|
||||||
|
### Option D: Fused multiply–accumulate (MAC) with Booth-3 (radix-8) and 4:2 compressors
|
||||||
|
- MUL: 3-cycle, also accepts an accumulate operand each cycle.
|
||||||
|
- Provides a useful primitive for dot-product kernels and soft-fp libraries.
|
||||||
|
- Divider: separate iterative non-restoring path.
|
||||||
|
- Area: similar to C, but the accumulator latch and bypass network add cost.
|
||||||
|
|
||||||
|
### Option E: Truncated 32×32→64 only, with MULHW/DIVW synthesized in microcode
|
||||||
|
- Illegal in general: software synthesis of MULH is too slow to be acceptable.
|
||||||
|
- Listed for completeness and to discard.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
| Option | MUL latency | MUL throughput | DIV latency (RV64) | Area (relative) | Design complexity |
|
||||||
|
|--------|-------------|----------------|---------------------|------------------|--------------------|
|
||||||
|
| A — Iterative Booth + NR div | 3–4 cyc | 1/cyc pipelined | 32–64 cyc | 0.6–0.8× | Low |
|
||||||
|
| B — Array + SRT-4 | 1 cyc | 1/cyc | 16 cyc | 2.5–3.0× | Medium (timing) |
|
||||||
|
| C — Wallace/Booth + SRT-4 | 3 cyc | 1/cyc | 16–20 cyc | 1.0–1.2× | High (tree) |
|
||||||
|
| D — MAC (Booth-3, 4:2) + NR div | 3 cyc | 1/cyc (with acc) | 32–64 cyc | 1.1–1.3× | High |
|
||||||
|
| E — Truncated, microcoded | n/a | n/a | n/a | smallest | n/a (incomplete) |
|
||||||
|
|
||||||
|
Numbers above are ORDER-OF-MAGNITUDE ESTIMATES based on published small-core implementations; INSUFFICIENT EVIDENCE exists to claim a specific gate count or µm² without synthesis at a known target node.
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
- Option A: smallest area per core → most 128-core replication headroom; trivially fits any pipeline depth; well-understood verification.
|
||||||
|
- Option B: best single-cycle latency; useful for OoO cores with tight issue windows.
|
||||||
|
- Option C: good balance of latency and area; widely used in commercial OoO cores.
|
||||||
|
- Option D: enables efficient dot-product and SIMD-style software; helps a soft FP stack.
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
- Option A: lower peak MUL throughput than pipelined trees; multi-cycle DIV may stall the issue queue on back-to-back DIVs.
|
||||||
|
- Option B: area is prohibitive at 128 cores; long critical path forces a slow core clock or deep pipelining (defeating the latency benefit).
|
||||||
|
- Option C: Wallace/Dadda partial-product reduction has irregular carry-save structures that are harder to formally verify and harder to fix in ECO.
|
||||||
|
- Option D: accumulator adds bypass and forwarding complexity into the pipeline and register-file writeback path; benefits only workloads that can be rewritten to use the MAC.
|
||||||
|
- Option E: not viable for an M-extension-compliant core.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
ASSUMPTION: XH-1 is a tiled, replicated design where per-core area is a first-class constraint because the whole 128-core array must fit in the package, power, and yield envelope. A small per-core MUL/DIV is therefore a high-value design point.
|
||||||
|
|
||||||
|
PROPOSAL: Treat MUL/DIV as one of the "shared-tile resource" candidates. Specifically:
|
||||||
|
- If 128 cores × 1 MUL/DIV per core exceeds the area budget, fall back to a per-cluster shared unit (e.g., 1 MUL/DIV per 4 or 8 cores) behind a dedicated interconnect port.
|
||||||
|
- The crossbar/coherence documents in the repository should be checked before committing to a per-core vs. shared decision.
|
||||||
|
- A shared unit complicates the scoreboard: the issue queue must track remote MUL/DIV latency, which can be 4–8× the per-core MUL latency.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
- Per-core area: Option A scales best; Option B is the worst case.
|
||||||
|
- Frequency: Option A and Option C both close timing at typical small-core targets; Option B is the only option likely to force a slower core clock.
|
||||||
|
- Yield: small per-core datapath → higher core yield, fewer fatal defects per die.
|
||||||
|
- Coherence traffic: a long-latency MUL/DIV keeps the core stalled but does not generate coherence traffic; an off-core MUL/DIV (shared) does, because the issuing core may continue past the result and the unit must return through the interconnect.
|
||||||
|
- Verification: per-core unit × 128 = 128 instances × one verification suite. This is a strong argument for the simplest microarchitecture that meets the latency target.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
- For most non-numeric workloads, MUL/DIV throughput does not bound IPC; latency does not either, because the operand is rarely on the critical path. A 3–4 cycle MUL and a 32–64 cycle DIV is acceptable.
|
||||||
|
- For numeric, crypto, and hash kernels, MUL throughput is critical; the pipelined-tree options (C, D) win by roughly 2–3× peak throughput.
|
||||||
|
- DIV is rarely on the hot path; a 32–64 cycle iterative divider is almost always sufficient.
|
||||||
|
|
||||||
|
RECOMMENDATION (conditional): If XH-1 targets general-purpose + occasional numeric, prefer Option A. If XH-1 targets numeric/CV/crypto explicitly, prefer Option C.
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
- A 64×64→128 array multiplier at a modern node is roughly the size of the register file's writeback port plus the ALU; this is the dominant per-core cost in Option B.
|
||||||
|
- A pipelined 3-stage tree (Option C) shrinks the per-stage critical path at the cost of three sets of partial-product reduction and accumulation latches.
|
||||||
|
- An iterative multiplier (Option A) is dominated by a single 33–64-bit adder and a shift register, and is the smallest.
|
||||||
|
- PROPOSAL: Capture the per-option area in the implementation document once the target node is fixed. Until then, treat the "relative" column in the comparison table as the working estimate.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
- Combinational array multipliers (Option B) toggle the entire partial-product array every cycle; energy per MUL is highest.
|
||||||
|
- Pipelined tree multipliers (Option C) distribute the switching across pipeline registers, lowering per-cycle peak power but with similar energy per operation.
|
||||||
|
- Iterative shift-and-add (Option A) is the lowest per-operation energy because the active datapath per cycle is small (one adder stage).
|
||||||
|
- 128-core replication: peak power is the product of per-core dynamic power and number of active cores. Option A keeps per-core power lowest, which is the most important axis for a 128-core envelope.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
- Sign handling: MULHSU requires signed×unsigned with full sign extension. The most common bug source in MUL/DIV is signed/unsigned mode selection; design the control path so that mode bits are sourced from the decoder, never from a sticky register.
|
||||||
|
- Zero-detection: DIV by zero must complete in bounded time and write a defined result. The IEEE / RISC-V rule is that DIV/REM by zero return −1 / the dividend. The unit must NOT trap on divide-by-zero; that is a software choice.
|
||||||
|
- Overflow: for DIV, the only overflow case is INT_MIN / −1. RISC-V mandates that this returns INT_MIN. The divider must detect this explicitly; iterative non-restoring dividers do, but SRT designs must be checked.
|
||||||
|
- Latency variability: DIV latency is data-dependent in non-restoring designs (it is fixed in SRT). If the pipeline assumes a fixed MUL/DIV latency, prefer an SRT divider or a constant-iteration iterative divider.
|
||||||
|
- Reset and scan: a 128-core replication multiplies the scan chain length. PROPOSAL: gate scan on the MUL/DIV unit to limit shift power, at the cost of reduced fault coverage. This is a verification trade-off.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
- The MUL/DIV unit has the highest ratio of corner cases to lines of RTL of any execution unit. Famous verification pitfalls:
|
||||||
|
- Signed multiplication: 0 × INT_MIN, INT_MIN × INT_MIN.
|
||||||
|
- MULH overflow into the high half with sign extension.
|
||||||
|
- MULHSU sign/unsigned mixing.
|
||||||
|
- DIV by zero, REM by zero.
|
||||||
|
- DIV overflow (INT_MIN / −1, INT_MIN % −1).
|
||||||
|
- All 16 combinations of signed/unsigned × 4 ops × 2 widths on RV64.
|
||||||
|
- PROPOSAL: Maintain a directed-test corpus at the unit level that exhausts the (sign × op × corner-operand) matrix, and a constrained-random suite at the core level.
|
||||||
|
- ASSUMPTION: XH-1 follows the riscv-formal convention of writing the MUL/DIV shadow model in a functional language. If so, the shadow model is the most expensive deliverable in the verification flow and is the strongest argument for picking the simplest microarchitecture.
|
||||||
|
|
||||||
|
## Software Considerations
|
||||||
|
- The compiler's latency model for MUL/DIV must match the hardware's; otherwise the scheduler will insert unnecessary stalls or miss scheduling opportunities. If Option A is chosen, the compiler should treat MUL as a 3–4 cycle latency op and DIV as a 32–64 cycle latency op.
|
||||||
|
- If MUL/DIV is moved off-core (shared), the toolchain needs an intrinsic or scheduling model that reflects the round-trip latency through the interconnect. This is non-trivial; most toolchains do not model non-uniform functional unit latency across cores.
|
||||||
|
- The Linux kernel's alternatives patching and the C library's soft-float paths sometimes use MUL/DIV in hot paths; verify that the chosen unit's latency is acceptable for the kernel configurations that the project intends to boot.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
PROPOSAL: For XH-1's first-pass implementation, adopt **Option A** (iterative Booth-2 multiplier pipelined to 1 result/cycle, non-restoring divider at 1 bit/cycle) for the following reasons:
|
||||||
|
1. Per-core area is minimized, which is the dominant axis in a 128-core replication.
|
||||||
|
2. The verification footprint is the smallest among the viable options.
|
||||||
|
3. The latency/throughput profile is acceptable for the majority of non-numeric workloads.
|
||||||
|
4. The microarchitecture is the easiest to ECO and to port across nodes.
|
||||||
|
|
||||||
|
PROPOSAL: If benchmarks (TBD) show that MUL/DIV throughput is on the critical path, migrate to **Option C** (pipelined 3-stage Booth/Wallace tree + SRT-4 divider) as a follow-on revision. Reserve **Option B** (fully combinational) for a hypothetical single-core "XH-1 Big" variant where replication is not the binding constraint.
|
||||||
|
|
||||||
|
ASSUMPTION: This recommendation is contingent on the pipeline depth, target clock, and per-core area budget not being fixed by other documents in the repository. If, for example, the pipeline document commits to a 5+ GHz target, Option A may not close timing and Option C becomes the floor rather than the upgrade.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
- Low-to-medium on the relative area numbers; they are order-of-magnitude estimates from published small-core work, not XH-1 synthesis.
|
||||||
|
- High on the qualitative trade-offs (area, verification complexity, replication cost).
|
||||||
|
- Low on the latency numbers, which depend on the target node and the adder architecture that is not yet specified in `alu.md` or `datapath.md`.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
- What is the target node and the target core clock? This gates the latency-vs-area decision.
|
||||||
|
- Is XH-1 RV32 or RV64? If RV32, the 32×32→64 multiplier is dramatically smaller.
|
||||||
|
- Is the M extension mandatory for the first tape-out, or can it be gated behind a config bit?
|
||||||
|
- Is there a plan for a shared per-cluster MAC/vector unit that subsumes the MUL/DIV? If so, the per-core MUL/DIV may be reduced to a minimal subset.
|
||||||
|
- What is the pipeline depth (see `pipeline.md`, currently SOON)? This determines whether a 1-cycle, 3-cycle, or N-cycle MUL fits cleanly into the execute stage.
|
||||||
|
- What is the issue model — in-order (single-issue, dual-issue, etc.)? The structural hazard model for MUL/DIV depends on it.
|
||||||
|
- Does the project plan to use `riscv-formal` for verification, or a UVM/constrained-random flow? The two flows have different optimal unit-level architectures.
|
||||||
|
- What are the integer workload targets — server, embedded, HPC, ML inference? Each implies a different MUL/DIV throughput requirement.
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
- The RISC-V "Unprivileged ISA" specification, M-extension chapter — INSUFFICIENT EVIDENCE to cite a specific revision without checking the repository. The relevant normative requirements are: MUL/MULH/MULHU/MULHSU produce full 2X-bit results, DIV/DIVU/REM/REMU round toward zero, DIV/REM by zero write back defined results, and INT_MIN / −1 is defined.
|
||||||
|
- Canonical microarchitecture textbooks (Hennessy & Patterson; Parhami; Ercegovac & Lang) — referenced only at the level of well-known technique names (Booth, Wallace, Dadda, SRT, non-restoring division). No specific page or claim is attributed.
|
||||||
|
- INSUFFICIENT EVIDENCE: any XH-1-specific synthesis, layout, or benchmark data. The comparison table is built from public-domain technique properties and must be re-validated with synthesis at the target node before tape-out.
|
||||||
+263
File diff suppressed because one or more lines are too long
@@ -0,0 +1,2 @@
|
|||||||
|
2026-08-25T18:13:01Z research/03-core-design/mul-div-unit.md 1 research completed
|
||||||
|
2026-08-25T18:13:16Z research/03-core-design/mul-div-unit.md 1 review api-failure
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
research/03-core-design/mul-div-unit.md
|
||||||
+248
@@ -0,0 +1,248 @@
|
|||||||
|
# XH-1 Multiply/Divide Unit Research
|
||||||
|
|
||||||
|
## Topic
|
||||||
|
|
||||||
|
`research/03-core-design/mul-div-unit.md` — design of the integer multiply/divide (MUL/DIV) execution unit for the XH-1 core, with particular attention to how this unit is replicated across 128 cores.
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
**Stub.** The current document contains only the placeholder `SOON`. This research document establishes the design-space analysis required to populate that stub. No committed micro-architecture exists yet.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
|
||||||
|
The integer multiply/divide unit (MDU) implements the RISC-V `M` extension (and optionally `B`, `K`, or vector variants) within each XH-1 core. This document surveys MDU micro-architectures — array multipliers, Wallace/Dadda trees, Booth-recoded multipliers, iterative shift-add multipliers, radix-2/radix-4/radix-8 dividers, and higher-radix/SRT dividers — and evaluates them against the XH-1 design constraints: per-core area and power budgets, latency targets, throughput requirements, verification complexity, and the multiplicative cost of replicating the unit across 128 cores. The analysis concludes that a parameterized iterative shift-add MDU with optional early-exit and a modest dedicated array multiplier is the most defensible starting point, but flags the assumption-laden nature of the recommendation.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
|
||||||
|
> What MDU micro-architecture best satisfies the XH-1 per-core area, power, latency, and throughput targets while remaining tractable to verify, to replicate 128 times, and to expose to the RISC-V ISA and toolchain?
|
||||||
|
|
||||||
|
Sub-questions:
|
||||||
|
|
||||||
|
1. Should multiplication and division share datapath hardware or be independent units?
|
||||||
|
2. What latency is required to avoid becoming a structural hazard in the XH-1 pipeline?
|
||||||
|
3. What is the minimum-radix divider that meets the per-core throughput target?
|
||||||
|
4. Does the cost of a fast array multiplier (e.g., radix-4 Booth with Wallace tree) justify its latency benefit over an iterative shift-add design replicated 128 times?
|
||||||
|
5. Which RISC-V extensions must be supported in v1 (`M` only vs. `M+B` vs. `M+B+K`)?
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
### RISC-V `M` Extension Semantics (FACT)
|
||||||
|
|
||||||
|
The RISC-V `M` extension defines the following operations on the `XLEN`-bit registers (RV64 assumed for XH-1 unless stated otherwise):
|
||||||
|
|
||||||
|
- `MUL`, `MULH`, `MULHSU`, `MULHU` — 64×64 → lower-64 / upper-64 signed/unsigned products.
|
||||||
|
- `DIV`, `DIVU`, `REM`, `REMU` — signed/unsigned division and remainder, 64÷64 → 64 quotient and 64 remainder.
|
||||||
|
- `MULW`, `DIVW`, `DIVUW`, `REMW`, `REMWU` — 32-bit operations that sign-extend the 32-bit result to 64 bits.
|
||||||
|
|
||||||
|
Corner cases the MDU must handle per the ISA:
|
||||||
|
|
||||||
|
- Division by zero returns `0xFFFFFFFFFFFFFFFF` (quotient) and the dividend (remainder) for unsigned, and `-1` / dividend for signed.
|
||||||
|
- The most-negative signed integer (`0x8000…`) divided by `-1` must return the dividend as quotient and `0` as remainder; this traps on x86 but is non-trapping in RISC-V.
|
||||||
|
- Overflow behavior is *only* defined for signed division/remainder by `-1`; no other overflow is architecturally visible.
|
||||||
|
|
||||||
|
### Latency vs. Throughput (FACT)
|
||||||
|
|
||||||
|
- A 64×64 → 128-bit full multiplication requires at minimum 128 partial-product rows for a non-Booth radix-2 design, ~64 rows for radix-4 Booth, ~43 rows for radix-8 Booth.
|
||||||
|
- Bit-serial shift-add multiplication completes in 64 cycles.
|
||||||
|
- Restoring division completes in 64 cycles; non-restoring in 32–64 cycles depending on the normalization of the remainder; higher-radix SRT dividers in O(log radix(N)) cycles but with substantially larger area.
|
||||||
|
|
||||||
|
### RISC-V `B` and `K` Extensions (FACT, partial)
|
||||||
|
|
||||||
|
- `B` (Bitmanip) under RV64 1.0.0 includes `MULH`, `MULHU`, `MULHSU` already in `M`, and adds rev-8.0/1.0 bitmanip groups such as `CLMUL`, `CLMULH`, `CLMULR`, `MIN[U]`, `MAX[U]`, `ANDN`, `ORN`, `XNOR`, `SEXT.B/H/W`, and the `Zba`/`Zbb`/`Zbs` sub-extensions.
|
||||||
|
- `K` (cryptography) adds `Zkn` (AES/SHA) and is not in scope for the base MDU.
|
||||||
|
- `Zmmul` provides multiply-only and is a common power-optimized subset of `M`.
|
||||||
|
|
||||||
|
## Existing Approaches
|
||||||
|
|
||||||
|
| Approach | Latency (typ., 64-bit) | Area (relative) | Throughput | Notes |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| Iterative shift-add multiplier | 64 cycles | 1× | 1 / 64 cycles | Smallest area; low throughput |
|
||||||
|
| Radix-4 Booth + carry-save | ~17 cycles | ~4–6× | 1 / ~17 | Common in embedded OoO |
|
||||||
|
| Radix-4 Booth + Wallace tree + final CPA | ~3–5 cycles | ~8–12× | 1 / cycle (pipelined) | Used in high-perf cores |
|
||||||
|
| Pipelined array multiplier (k stages) | k cycles | large | 1 / cycle | Throughput-bound, area-heavy |
|
||||||
|
| Radix-2 non-restoring divider | 32–64 cycles | 1× (shared with MUL possible) | 1 / 32–64 | Low area |
|
||||||
|
| Radix-4 SRT divider | ~16 cycles | ~3–5× | 1 / ~16 | Moderate |
|
||||||
|
| Radix-8/16 SRT divider | ~8–10 cycles | large | 1 / ~10 | High-end OoO |
|
||||||
|
| Combinational 64×64 array (no pipelining) | ~10+ gate delays | very large | 1 / many cycles | Impractical |
|
||||||
|
| Shared MUL/DIV datapath (Liang/Montusiewicz style) | variable | ~1.5× single-function | variable | Saves ~30–40% area at the cost of CPI |
|
||||||
|
|
||||||
|
(Quantitative magnitudes in this table are estimates from public micro-architecture literature; values vary considerably with cell library, target frequency, and pipeline integration. See Sources.)
|
||||||
|
|
||||||
|
## Alternative Designs
|
||||||
|
|
||||||
|
### A. Pure Iterative Shift-Add MDU
|
||||||
|
|
||||||
|
A single 128-bit accumulator-based datapath. Each cycle shifts the multiplicand and adds if the corresponding multiplier bit is set, or performs a non-restoring division step. The same hardware services both MUL and DIV.
|
||||||
|
|
||||||
|
### B. Pipelined Iterative MDU
|
||||||
|
|
||||||
|
Same datapath as A, but cut into N pipeline stages (commonly 2 or 4) to allow clock-frequency scaling. Issues per-cycle throughput remains 1 op / N cycles for a single operand width.
|
||||||
|
|
||||||
|
### C. Dedicated Array Multiplier + Iterative Divider
|
||||||
|
|
||||||
|
A small radix-4 Booth/Wallace multiplier (lower 64 bits in 3–4 cycles) and a separate radix-2 or radix-4 non-restoring divider. Two independent units that can operate concurrently.
|
||||||
|
|
||||||
|
### D. Shared Pipelined Radix-4 MDU
|
||||||
|
|
||||||
|
A single radix-4 Booth partial-product generator feeding a shared CSA tree, with a reconfigurable final-stage adder and a small division state machine. Latency: ~5–8 cycles for MUL, ~16 for DIV.
|
||||||
|
|
||||||
|
### E. Pipelined High-Performance MDU (Wallace/Wallace + SRT-4)
|
||||||
|
|
||||||
|
Multi-cycle but pipelined radix-4 multiplier and SRT-4 divider. Both deliver 1 op/cycle throughput once steady-state. Area dominates the integer datapath.
|
||||||
|
|
||||||
|
### F. `Zmmul`-Only MUL, External DIV Trap
|
||||||
|
|
||||||
|
Implement MUL in hardware (small array or iterative), trap DIV/REM to a software handler. The privileged trap path is fast on XH-1 only if the software model tolerates it; SPECint and most server workloads execute division frequently enough that this is usually unacceptable.
|
||||||
|
|
||||||
|
### G. Compressed / Shared Across Cores
|
||||||
|
|
||||||
|
A single physical MDU shared by N cores via a NoC-side arbiter. Removes per-core area cost but introduces structural contention; this is generally considered an *adverse* design for a 128-core tiled machine.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
| Design | Area/cell count (est.) | Cycles/op MUL | Cycles/op DIV | Throughput (steady) | Verif. complexity | Replicated ×128 cost | Power/cores at 1 GHz (est.) |
|
||||||
|
|---|---|---|---|---|---|---|---|
|
||||||
|
| A. Iterative | 1.0× | 64 | 32–64 | 1/64 | Low | low | low |
|
||||||
|
| B. Pipelined iter. (2-stage) | ~1.1× | 32 | 16–32 | 1/32 | Low–Medium | low | low–medium |
|
||||||
|
| C. Array MUL + iter DIV | ~6–8× | 3–5 | 32–64 | 1/cycle (MUL) | Medium | high | medium–high |
|
||||||
|
| D. Shared radix-4 | ~4–6× | 5–8 | 16 | 1/cycle (MUL) | Medium–High | medium | medium |
|
||||||
|
| E. Pipelined Wallace + SRT-4 | ~12–20× | 1/cycle | 1/cycle | 1/cycle | High | very high | high |
|
||||||
|
| F. Zmmul + trap | ~1–2× | variable | trap | low | Lowest | lowest | lowest |
|
||||||
|
| G. Shared across cores | 1/N per core | +NoC latency | +NoC latency | contended | High (coherency) | lowest in area, highest in latency | lowest in static power, high in dynamic on miss |
|
||||||
|
|
||||||
|
(All non-FACT values are PROPOSAL/ESTIMATE pending the XH-1 cell library and frequency target.)
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
|
||||||
|
- **A (iterative):** lowest area and verification cost; trivially replicated ×128; deterministic timing; well-suited to in-order pipelines with a single MDU reservation station.
|
||||||
|
- **C (array + iter):** low MUL latency benefits compiled code with tight multiply chains (CRC, hashing, address computation); the iterative divider keeps area under control.
|
||||||
|
- **D (shared radix-4):** good balance; single unit means fewer ports on the issue queue, simpler back-pressure, easier to verify than E.
|
||||||
|
- **E (pipelined):** removes MDU as a critical path for high-frequency operation; appropriate only if XH-1 is a high-frequency out-of-order core.
|
||||||
|
- **F (Zmmul + trap):** smallest possible die; useful if division is genuinely rare on the target workload.
|
||||||
|
- **G (shared across cores):** minimal replicated area; only viable if the NoC has spare bandwidth and the workload tolerates MUL/DIV latency (uncommon).
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
|
||||||
|
- **A:** 64-cycle MUL is a structural hazard in nearly all non-trivial programs. Most RISC-V cores ship a faster multiplier.
|
||||||
|
- **B:** still poor MUL throughput; the pipelining helps frequency but not CPI for multiply-heavy code.
|
||||||
|
- **C:** the array multiplier is the largest single addition; replicating 128 instances dominates the per-core MDU area. The iterative divider becomes the long pole.
|
||||||
|
- **D:** the radix-4 booth encoder + Wallace reduction is the single most verification-intensive block in the integer datapath; high combinatorial depth impacts timing closure.
|
||||||
|
- **E:** the worst replication cost; if the XH-1 core is in-order (not established by repository context), this is overkill.
|
||||||
|
- **F:** division traps in scientific, crypto, and some HPC kernels; would require a fast software path that defeats the area saving.
|
||||||
|
- **G:** a single shared MDU becomes a hot spot; under load, the queuing delay exceeds a 64-cycle per-core iterative design. Also violates the principle that each core should be self-sufficient for any single instruction.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
|
||||||
|
- **ISA scope (ASSUMPTION):** XH-1 implements RV64IMAC or RV64GC. Until the decoder document (`research/03-core-design/decoder.md`) defines the supported extension set, MDU scope must include both `M` and the bitmanip multiply subset if `B` is present.
|
||||||
|
- **Pipeline style (UNKNOWN):** if the XH-1 core is in-order, an iterative or low-radix MDU is sufficient. If out-of-order, a pipelined unit is more defensible. The repository does not yet establish this.
|
||||||
|
- **Per-core MDU ports (ASSUMPTION):** at most one MDU op per cycle per core, regardless of out-of-order issue width.
|
||||||
|
- **FPU integration (OPEN):** RISC-V `F`/`D` extensions can be serviced by a separate FPU MDU (for fused multiply-add) or by forwarding to the integer MDU. Coupling decisions affect the MDU port count.
|
||||||
|
- **NoC and coherence (ASSUMPTION):** the 128 cores share a coherent memory subsystem; the MDU reads two source registers and writes one (or two for `MULH`), so register-file read/write port count is the binding constraint, not the NoC.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
|
||||||
|
- The replication cost of any MDU is *exactly* the per-core area × 128 plus shared overhead. A design choice that adds 0.05 mm² per core adds **6.4 mm²** across the chip. The C/D/E rows of the comparison table are therefore the most consequential decision in the MDU document.
|
||||||
|
- A shared-across-cores MDU (G) trades replicated area for **structural contention** at 128 cores. With a single shared unit, even a steady-state mix of 10% multiply/divide instructions causes queueing proportional to (1 / (1 − utilization)). At 50% utilization, mean wait time already exceeds the latency of an iterative per-core design. (Estimate, Little's law, INSUFFICIENT EVIDENCE for absolute numbers without a workload profile.)
|
||||||
|
- A *clustered* variant — one MDU per cluster of 4 or 8 cores — reduces the replication cost while bounding the worst-case contention. This is RECOMMENDATION-pending and is treated as an OPEN QUESTION below.
|
||||||
|
- The MDU contributes directly to per-core power density; if XH-1 is thermally constrained at 128 cores, an iterative design (A/B) is materially preferable.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
|
||||||
|
- A 64-cycle MUL means a `for (i=0;i<n;i++) h = h * 31 + buf[i];` style loop runs at ~64 cycles per byte hashed. A 4-cycle pipelined MUL runs at ~4 cycles per byte. Workloads with high multiply density (cryptography, signal processing, JIT compilers, regex, hash joins) are order-of-magnitude sensitive to this number.
|
||||||
|
- A 64-cycle DIV is acceptable for most scalar code; high-radix SRT dividers benefit primarily vectorized divide loops. INSUFFICIENT EVIDENCE to assume XH-1 will see heavy scalar divide.
|
||||||
|
- The MDU rarely becomes the *bottleneck* even at 64 cycles/iter, because most programs do not back-to-back issue multiplies. The MDU *does* become a critical path on a small set of inner loops.
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
|
||||||
|
Per-order estimates for a modern 7nm-class cell library (ASSUMPTION — no XH-1 process node established):
|
||||||
|
|
||||||
|
- Iterative 64-bit shift-add: ~3,000–6,000 gates.
|
||||||
|
- Radix-4 Booth + Wallace + CPA, 64-bit: ~15,000–30,000 gates.
|
||||||
|
- Pipelined (3-stage) radix-4 64-bit: ~25,000–45,000 gates.
|
||||||
|
- SRT-4 divider: ~10,000–20,000 gates.
|
||||||
|
- These are PROPOSAL figures pending a real synthesis. They vary ±2× with library and timing constraints.
|
||||||
|
|
||||||
|
Multiplying by 128:
|
||||||
|
|
||||||
|
- 5,000 gates × 128 = 640,000 gates (~0.3–0.5 mm² at 7 nm, est.).
|
||||||
|
- 30,000 gates × 128 = 3.84 M gates (~2.5–4 mm² est.).
|
||||||
|
- 50,000 gates × 128 = 6.4 M gates (~4–7 mm² est.).
|
||||||
|
|
||||||
|
A full XH-1 die estimate is INSUFFICIENT EVIDENCE without a floorplan.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
|
||||||
|
- Combinational depth × switching activity × capacitance determines dynamic power. Iterative shift-add has the lowest dynamic power at the cost of executing more cycles; a fully combinational array multiplier has the highest dynamic power per op.
|
||||||
|
- Per-op energy is a *U-shaped* function of latency: very short-latency units burn more energy per cycle but fewer cycles; very long-latency units have low per-cycle leakage but high total leakage over 64 cycles. The minimum is typically in the 4–16 cycle range for a 64-bit MUL. (PROPOSAL — general principle, not specific to XH-1.)
|
||||||
|
- Replicating a high-power MDU 128 times has thermal and PDN implications: the power distribution network must support the worst-case aggregate.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
|
||||||
|
- Booth encoders, partial-product compressors, and SRT quotient-selection logic are notoriously timing-sensitive. Synthesis with `set_max_delay -datapath_only` is generally required; the floorplan must hold these cells together.
|
||||||
|
- Iterative shift-add trivially implements in standard cell logic; no special datapath cell library is required.
|
||||||
|
- Sharing multiplier and divider datapath (Design D) requires careful multiplexing that introduces a critical-path hazard; many designs simply keep them as separate functional units to ease timing closure.
|
||||||
|
- The MDU corner-case logic (divide-by-zero, signed-overflow-on-(-1)) is small but error-prone; reference model should be exhaustive on these inputs in verification.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
|
||||||
|
- Iterative shift-add: the state space is small; UVM directed and constrained-random coverage is straightforward.
|
||||||
|
- Array multiplier: corner cases (e.g., operand = `0x8000_0000_0000_0000` in `MULHSU`, mixed sign/zero) require formal verification of the partial-product array. Strongly recommend a co-simulation against a software reference (e.g., the Spike golden model or a Python oracle).
|
||||||
|
- SRT divider: quotient-digit selection is the standard formal-verification target; the choice of *r* (radix), *p* (redundancy), and the PLA/ROM selection table is sensitive to off-by-one in the redundancy constants. Several published bugs in commercial processors stem from SRT tables.
|
||||||
|
- Across 128 cores, *manufacturing* test coverage becomes a concern: each MDU must be DC/AC testable, requiring scan insertion and possibly BIST.
|
||||||
|
- Cross-bar functional verification (does each core's MDU behave identically?) is most easily done with identical replicated RTL — a heterogeneous MDU design increases verification surface.
|
||||||
|
|
||||||
|
## Software Considerations
|
||||||
|
|
||||||
|
- Compilers generate `MUL`/`MULH` pairs for 128-bit multiplications. If MUL is fast but MULH is slow (or vice versa), the *paired* latency matters, not the per-instruction latency.
|
||||||
|
- `Zmmul` (no DIV) is rarely enabled in general-purpose code because GCC/Clang emit DIV freely; turning it on requires recompiling with `-mno-div` and accepting performance loss on division-heavy code.
|
||||||
|
- The RISC-V psABI requires MUL/DIV results in the integer register file; the MDU must write the integer RF, not a sidecar.
|
||||||
|
- Vector (`V`) and Packed-SIMD (`P`) extensions, if added later, may have separate vector MDU requirements that are *out of scope* for this document.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
|
||||||
|
**RECOMMENDATION (provisional, low-to-medium confidence):** Adopt **Design B (pipelined iterative MDU)** as the v1 default, with the data path sized to support the `M` extension. Rationale:
|
||||||
|
|
||||||
|
- 128× replication makes the per-core area of Designs C/D/E potentially prohibitive (PROPOSAL — depends on the die area budget, which is INSUFFICIENT EVIDENCE).
|
||||||
|
- A 2-stage pipelined iterative design provides ~32 cycles/op MUL and ~16 cycles/op DIV, which is sufficient for the majority of scalar RISC-V code paths and substantially cheaper to verify than any array/SRT design.
|
||||||
|
- A small (optional) **Design C fallback** — a dedicated 64×64 → 64 lower radix-4 array multiplier alongside the iterative divider — should be retained as a synthesis experiment in early RTL exploration to confirm the area/power cost is acceptable.
|
||||||
|
- If early benchmarks (INSUFFICIENT EVIDENCE) show MUL throughput as a binding constraint, escalate to Design D; if the core is confirmed in-order, Design A may be reconsidered.
|
||||||
|
|
||||||
|
This recommendation is explicitly contingent on the unresolved items in **Open Questions** and should be revisited when the pipeline and decoder documents are populated.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
|
||||||
|
| Sub-area | Confidence | Reason |
|
||||||
|
|---|---|---|
|
||||||
|
| RISC-V `M` semantics | **High** | Spec is unambiguous and well-established. |
|
||||||
|
| Latency of iterative vs. array | **High** | Standard textbook numbers; broadly verified. |
|
||||||
|
| Per-core area estimates | **Low** | No process node, no cell library, no frequency target. |
|
||||||
|
| 128-core replication impact | **Low–Medium** | Magnitudes are defensible, but absolute numbers depend on die budget and floorplan. |
|
||||||
|
| Verifier complexity ranking | **Medium** | General industry consensus; no XH-1 verification plan exists. |
|
||||||
|
| Pipeline style (in-order vs. OoO) | **None** | Not established by repository context; this single unknown dominates the recommendation. |
|
||||||
|
| Per-core MDU port count | **Low** | Not established. |
|
||||||
|
| FPU / vector / bitmanip scope | **Low** | Awaiting decoder document. |
|
||||||
|
|
||||||
|
Overall recommendation confidence: **Low-to-Medium.** The recommendation is the least-bad default under the current evidence; it should not be treated as committed.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
1. **Is the XH-1 core in-order or out-of-order?** (Blocks the latency target and the throughput requirement.)
|
||||||
|
2. **What is the target clock frequency and process node?** (Determines whether a radix-4 array multiplier can meet timing in a single cycle.)
|
||||||
|
3. **What is the per-core die-area budget for the integer execution units?** (Drives the 128× replication cost.)
|
||||||
|
4. **What is the supported ISA extension set in v1?** (M only, IMAC, GC, or GCB+V?)
|
||||||
|
5. **Does the FPU have its own multiplier, or does it forward to the integer MDU?** (Affects port count and result-bus topology.)
|
||||||
|
6. **What is the expected workload mix on XH-1?** (Server, HPC, embedded, AI inference — different mix radically changes the MUL/DIV latency sensitivity.)
|
||||||
|
7. **Is a clustered MDU (one per 4 or 8 cores) an acceptable design point?** (Requires coherency/NoC discussion, currently unaddressed in this document tree.)
|
||||||
|
8. **Is the MDU required to be IEEE-754 compatible in any way, or only RISC-V `M`/`B` semantics?** (FPU integration question.)
|
||||||
|
9. **What is the verification methodology — UVM, formal, both, lightweight directed-random?** (Drives the complexity analysis above.)
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
- RISC-V *Unprivileged ISA Specification*, Volume I, Document Version 20191213 (and later). Publicly available at the RISC-V International site. Used for `M`/`B` extension semantics and corner cases.
|
||||||
|
- RISC-V *“B” Bitmanip Extension*, version 1.0.0 draft (RISC-V International). Used for the bitmanip scope discussion.
|
||||||
|
- Hennessy & Patterson, *Computer Architecture: A Quantitative Approach* (multiple editions) — general background on multiplier/divider design and area/energy tradeoffs. Specific quantitative claims in this document are not from this source and are labelled ESTIMATE/PROPOSAL.
|
||||||
|
- Weste & Harris, *CMOS VLSI Design: A Circuits and Systems Perspective* — array multiplier, Booth, and Wallace tree area/latency discussion. Magnitudes in the comparison table are not directly cited from this text; they are PROPOSAL/ESTIMATE.
|
||||||
|
- Ercegovac & Lang, *Digital Arithmetic* — SRT division, quotient-digit selection tables. Specific radix-4 selection tables are not reproduced here.
|
||||||
|
- *The Microarchitecture of Pipeline and Superscalar CPUs* (Shen, Lipasti) — pipelined iterative vs. array multiplier tradeoffs. General principles only.
|
||||||
|
|
||||||
|
INSUFFICIENT EVIDENCE: any XH-1-specific synthesis result, benchmark number, die area, frequency target, or workload measurement. All numerical values in this document that are not directly from the RISC-V spec are PROPOSAL or ESTIMATE.
|
||||||
+248
@@ -0,0 +1,248 @@
|
|||||||
|
# XH-1 Multiply/Divide Unit Research
|
||||||
|
|
||||||
|
## Topic
|
||||||
|
|
||||||
|
`research/03-core-design/mul-div-unit.md` — design of the integer multiply/divide (MUL/DIV) execution unit for the XH-1 core, with particular attention to how this unit is replicated across 128 cores.
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
**Stub.** The current document contains only the placeholder `SOON`. This research document establishes the design-space analysis required to populate that stub. No committed micro-architecture exists yet.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
|
||||||
|
The integer multiply/divide unit (MDU) implements the RISC-V `M` extension (and optionally `B`, `K`, or vector variants) within each XH-1 core. This document surveys MDU micro-architectures — array multipliers, Wallace/Dadda trees, Booth-recoded multipliers, iterative shift-add multipliers, radix-2/radix-4/radix-8 dividers, and higher-radix/SRT dividers — and evaluates them against the XH-1 design constraints: per-core area and power budgets, latency targets, throughput requirements, verification complexity, and the multiplicative cost of replicating the unit across 128 cores. The analysis concludes that a parameterized iterative shift-add MDU with optional early-exit and a modest dedicated array multiplier is the most defensible starting point, but flags the assumption-laden nature of the recommendation.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
|
||||||
|
> What MDU micro-architecture best satisfies the XH-1 per-core area, power, latency, and throughput targets while remaining tractable to verify, to replicate 128 times, and to expose to the RISC-V ISA and toolchain?
|
||||||
|
|
||||||
|
Sub-questions:
|
||||||
|
|
||||||
|
1. Should multiplication and division share datapath hardware or be independent units?
|
||||||
|
2. What latency is required to avoid becoming a structural hazard in the XH-1 pipeline?
|
||||||
|
3. What is the minimum-radix divider that meets the per-core throughput target?
|
||||||
|
4. Does the cost of a fast array multiplier (e.g., radix-4 Booth with Wallace tree) justify its latency benefit over an iterative shift-add design replicated 128 times?
|
||||||
|
5. Which RISC-V extensions must be supported in v1 (`M` only vs. `M+B` vs. `M+B+K`)?
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
### RISC-V `M` Extension Semantics (FACT)
|
||||||
|
|
||||||
|
The RISC-V `M` extension defines the following operations on the `XLEN`-bit registers (RV64 assumed for XH-1 unless stated otherwise):
|
||||||
|
|
||||||
|
- `MUL`, `MULH`, `MULHSU`, `MULHU` — 64×64 → lower-64 / upper-64 signed/unsigned products.
|
||||||
|
- `DIV`, `DIVU`, `REM`, `REMU` — signed/unsigned division and remainder, 64÷64 → 64 quotient and 64 remainder.
|
||||||
|
- `MULW`, `DIVW`, `DIVUW`, `REMW`, `REMWU` — 32-bit operations that sign-extend the 32-bit result to 64 bits.
|
||||||
|
|
||||||
|
Corner cases the MDU must handle per the ISA:
|
||||||
|
|
||||||
|
- Division by zero returns `0xFFFFFFFFFFFFFFFF` (quotient) and the dividend (remainder) for unsigned, and `-1` / dividend for signed.
|
||||||
|
- The most-negative signed integer (`0x8000…`) divided by `-1` must return the dividend as quotient and `0` as remainder; this traps on x86 but is non-trapping in RISC-V.
|
||||||
|
- Overflow behavior is *only* defined for signed division/remainder by `-1`; no other overflow is architecturally visible.
|
||||||
|
|
||||||
|
### Latency vs. Throughput (FACT)
|
||||||
|
|
||||||
|
- A 64×64 → 128-bit full multiplication requires at minimum 128 partial-product rows for a non-Booth radix-2 design, ~64 rows for radix-4 Booth, ~43 rows for radix-8 Booth.
|
||||||
|
- Bit-serial shift-add multiplication completes in 64 cycles.
|
||||||
|
- Restoring division completes in 64 cycles; non-restoring in 32–64 cycles depending on the normalization of the remainder; higher-radix SRT dividers in O(log radix(N)) cycles but with substantially larger area.
|
||||||
|
|
||||||
|
### RISC-V `B` and `K` Extensions (FACT, partial)
|
||||||
|
|
||||||
|
- `B` (Bitmanip) under RV64 1.0.0 includes `MULH`, `MULHU`, `MULHSU` already in `M`, and adds rev-8.0/1.0 bitmanip groups such as `CLMUL`, `CLMULH`, `CLMULR`, `MIN[U]`, `MAX[U]`, `ANDN`, `ORN`, `XNOR`, `SEXT.B/H/W`, and the `Zba`/`Zbb`/`Zbs` sub-extensions.
|
||||||
|
- `K` (cryptography) adds `Zkn` (AES/SHA) and is not in scope for the base MDU.
|
||||||
|
- `Zmmul` provides multiply-only and is a common power-optimized subset of `M`.
|
||||||
|
|
||||||
|
## Existing Approaches
|
||||||
|
|
||||||
|
| Approach | Latency (typ., 64-bit) | Area (relative) | Throughput | Notes |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| Iterative shift-add multiplier | 64 cycles | 1× | 1 / 64 cycles | Smallest area; low throughput |
|
||||||
|
| Radix-4 Booth + carry-save | ~17 cycles | ~4–6× | 1 / ~17 | Common in embedded OoO |
|
||||||
|
| Radix-4 Booth + Wallace tree + final CPA | ~3–5 cycles | ~8–12× | 1 / cycle (pipelined) | Used in high-perf cores |
|
||||||
|
| Pipelined array multiplier (k stages) | k cycles | large | 1 / cycle | Throughput-bound, area-heavy |
|
||||||
|
| Radix-2 non-restoring divider | 32–64 cycles | 1× (shared with MUL possible) | 1 / 32–64 | Low area |
|
||||||
|
| Radix-4 SRT divider | ~16 cycles | ~3–5× | 1 / ~16 | Moderate |
|
||||||
|
| Radix-8/16 SRT divider | ~8–10 cycles | large | 1 / ~10 | High-end OoO |
|
||||||
|
| Combinational 64×64 array (no pipelining) | ~10+ gate delays | very large | 1 / many cycles | Impractical |
|
||||||
|
| Shared MUL/DIV datapath (Liang/Montusiewicz style) | variable | ~1.5× single-function | variable | Saves ~30–40% area at the cost of CPI |
|
||||||
|
|
||||||
|
(Quantitative magnitudes in this table are estimates from public micro-architecture literature; values vary considerably with cell library, target frequency, and pipeline integration. See Sources.)
|
||||||
|
|
||||||
|
## Alternative Designs
|
||||||
|
|
||||||
|
### A. Pure Iterative Shift-Add MDU
|
||||||
|
|
||||||
|
A single 128-bit accumulator-based datapath. Each cycle shifts the multiplicand and adds if the corresponding multiplier bit is set, or performs a non-restoring division step. The same hardware services both MUL and DIV.
|
||||||
|
|
||||||
|
### B. Pipelined Iterative MDU
|
||||||
|
|
||||||
|
Same datapath as A, but cut into N pipeline stages (commonly 2 or 4) to allow clock-frequency scaling. Issues per-cycle throughput remains 1 op / N cycles for a single operand width.
|
||||||
|
|
||||||
|
### C. Dedicated Array Multiplier + Iterative Divider
|
||||||
|
|
||||||
|
A small radix-4 Booth/Wallace multiplier (lower 64 bits in 3–4 cycles) and a separate radix-2 or radix-4 non-restoring divider. Two independent units that can operate concurrently.
|
||||||
|
|
||||||
|
### D. Shared Pipelined Radix-4 MDU
|
||||||
|
|
||||||
|
A single radix-4 Booth partial-product generator feeding a shared CSA tree, with a reconfigurable final-stage adder and a small division state machine. Latency: ~5–8 cycles for MUL, ~16 for DIV.
|
||||||
|
|
||||||
|
### E. Pipelined High-Performance MDU (Wallace/Wallace + SRT-4)
|
||||||
|
|
||||||
|
Multi-cycle but pipelined radix-4 multiplier and SRT-4 divider. Both deliver 1 op/cycle throughput once steady-state. Area dominates the integer datapath.
|
||||||
|
|
||||||
|
### F. `Zmmul`-Only MUL, External DIV Trap
|
||||||
|
|
||||||
|
Implement MUL in hardware (small array or iterative), trap DIV/REM to a software handler. The privileged trap path is fast on XH-1 only if the software model tolerates it; SPECint and most server workloads execute division frequently enough that this is usually unacceptable.
|
||||||
|
|
||||||
|
### G. Compressed / Shared Across Cores
|
||||||
|
|
||||||
|
A single physical MDU shared by N cores via a NoC-side arbiter. Removes per-core area cost but introduces structural contention; this is generally considered an *adverse* design for a 128-core tiled machine.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
| Design | Area/cell count (est.) | Cycles/op MUL | Cycles/op DIV | Throughput (steady) | Verif. complexity | Replicated ×128 cost | Power/cores at 1 GHz (est.) |
|
||||||
|
|---|---|---|---|---|---|---|---|
|
||||||
|
| A. Iterative | 1.0× | 64 | 32–64 | 1/64 | Low | low | low |
|
||||||
|
| B. Pipelined iter. (2-stage) | ~1.1× | 32 | 16–32 | 1/32 | Low–Medium | low | low–medium |
|
||||||
|
| C. Array MUL + iter DIV | ~6–8× | 3–5 | 32–64 | 1/cycle (MUL) | Medium | high | medium–high |
|
||||||
|
| D. Shared radix-4 | ~4–6× | 5–8 | 16 | 1/cycle (MUL) | Medium–High | medium | medium |
|
||||||
|
| E. Pipelined Wallace + SRT-4 | ~12–20× | 1/cycle | 1/cycle | 1/cycle | High | very high | high |
|
||||||
|
| F. Zmmul + trap | ~1–2× | variable | trap | low | Lowest | lowest | lowest |
|
||||||
|
| G. Shared across cores | 1/N per core | +NoC latency | +NoC latency | contended | High (coherency) | lowest in area, highest in latency | lowest in static power, high in dynamic on miss |
|
||||||
|
|
||||||
|
(All non-FACT values are PROPOSAL/ESTIMATE pending the XH-1 cell library and frequency target.)
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
|
||||||
|
- **A (iterative):** lowest area and verification cost; trivially replicated ×128; deterministic timing; well-suited to in-order pipelines with a single MDU reservation station.
|
||||||
|
- **C (array + iter):** low MUL latency benefits compiled code with tight multiply chains (CRC, hashing, address computation); the iterative divider keeps area under control.
|
||||||
|
- **D (shared radix-4):** good balance; single unit means fewer ports on the issue queue, simpler back-pressure, easier to verify than E.
|
||||||
|
- **E (pipelined):** removes MDU as a critical path for high-frequency operation; appropriate only if XH-1 is a high-frequency out-of-order core.
|
||||||
|
- **F (Zmmul + trap):** smallest possible die; useful if division is genuinely rare on the target workload.
|
||||||
|
- **G (shared across cores):** minimal replicated area; only viable if the NoC has spare bandwidth and the workload tolerates MUL/DIV latency (uncommon).
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
|
||||||
|
- **A:** 64-cycle MUL is a structural hazard in nearly all non-trivial programs. Most RISC-V cores ship a faster multiplier.
|
||||||
|
- **B:** still poor MUL throughput; the pipelining helps frequency but not CPI for multiply-heavy code.
|
||||||
|
- **C:** the array multiplier is the largest single addition; replicating 128 instances dominates the per-core MDU area. The iterative divider becomes the long pole.
|
||||||
|
- **D:** the radix-4 booth encoder + Wallace reduction is the single most verification-intensive block in the integer datapath; high combinatorial depth impacts timing closure.
|
||||||
|
- **E:** the worst replication cost; if the XH-1 core is in-order (not established by repository context), this is overkill.
|
||||||
|
- **F:** division traps in scientific, crypto, and some HPC kernels; would require a fast software path that defeats the area saving.
|
||||||
|
- **G:** a single shared MDU becomes a hot spot; under load, the queuing delay exceeds a 64-cycle per-core iterative design. Also violates the principle that each core should be self-sufficient for any single instruction.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
|
||||||
|
- **ISA scope (ASSUMPTION):** XH-1 implements RV64IMAC or RV64GC. Until the decoder document (`research/03-core-design/decoder.md`) defines the supported extension set, MDU scope must include both `M` and the bitmanip multiply subset if `B` is present.
|
||||||
|
- **Pipeline style (UNKNOWN):** if the XH-1 core is in-order, an iterative or low-radix MDU is sufficient. If out-of-order, a pipelined unit is more defensible. The repository does not yet establish this.
|
||||||
|
- **Per-core MDU ports (ASSUMPTION):** at most one MDU op per cycle per core, regardless of out-of-order issue width.
|
||||||
|
- **FPU integration (OPEN):** RISC-V `F`/`D` extensions can be serviced by a separate FPU MDU (for fused multiply-add) or by forwarding to the integer MDU. Coupling decisions affect the MDU port count.
|
||||||
|
- **NoC and coherence (ASSUMPTION):** the 128 cores share a coherent memory subsystem; the MDU reads two source registers and writes one (or two for `MULH`), so register-file read/write port count is the binding constraint, not the NoC.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
|
||||||
|
- The replication cost of any MDU is *exactly* the per-core area × 128 plus shared overhead. A design choice that adds 0.05 mm² per core adds **6.4 mm²** across the chip. The C/D/E rows of the comparison table are therefore the most consequential decision in the MDU document.
|
||||||
|
- A shared-across-cores MDU (G) trades replicated area for **structural contention** at 128 cores. With a single shared unit, even a steady-state mix of 10% multiply/divide instructions causes queueing proportional to (1 / (1 − utilization)). At 50% utilization, mean wait time already exceeds the latency of an iterative per-core design. (Estimate, Little's law, INSUFFICIENT EVIDENCE for absolute numbers without a workload profile.)
|
||||||
|
- A *clustered* variant — one MDU per cluster of 4 or 8 cores — reduces the replication cost while bounding the worst-case contention. This is RECOMMENDATION-pending and is treated as an OPEN QUESTION below.
|
||||||
|
- The MDU contributes directly to per-core power density; if XH-1 is thermally constrained at 128 cores, an iterative design (A/B) is materially preferable.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
|
||||||
|
- A 64-cycle MUL means a `for (i=0;i<n;i++) h = h * 31 + buf[i];` style loop runs at ~64 cycles per byte hashed. A 4-cycle pipelined MUL runs at ~4 cycles per byte. Workloads with high multiply density (cryptography, signal processing, JIT compilers, regex, hash joins) are order-of-magnitude sensitive to this number.
|
||||||
|
- A 64-cycle DIV is acceptable for most scalar code; high-radix SRT dividers benefit primarily vectorized divide loops. INSUFFICIENT EVIDENCE to assume XH-1 will see heavy scalar divide.
|
||||||
|
- The MDU rarely becomes the *bottleneck* even at 64 cycles/iter, because most programs do not back-to-back issue multiplies. The MDU *does* become a critical path on a small set of inner loops.
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
|
||||||
|
Per-order estimates for a modern 7nm-class cell library (ASSUMPTION — no XH-1 process node established):
|
||||||
|
|
||||||
|
- Iterative 64-bit shift-add: ~3,000–6,000 gates.
|
||||||
|
- Radix-4 Booth + Wallace + CPA, 64-bit: ~15,000–30,000 gates.
|
||||||
|
- Pipelined (3-stage) radix-4 64-bit: ~25,000–45,000 gates.
|
||||||
|
- SRT-4 divider: ~10,000–20,000 gates.
|
||||||
|
- These are PROPOSAL figures pending a real synthesis. They vary ±2× with library and timing constraints.
|
||||||
|
|
||||||
|
Multiplying by 128:
|
||||||
|
|
||||||
|
- 5,000 gates × 128 = 640,000 gates (~0.3–0.5 mm² at 7 nm, est.).
|
||||||
|
- 30,000 gates × 128 = 3.84 M gates (~2.5–4 mm² est.).
|
||||||
|
- 50,000 gates × 128 = 6.4 M gates (~4–7 mm² est.).
|
||||||
|
|
||||||
|
A full XH-1 die estimate is INSUFFICIENT EVIDENCE without a floorplan.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
|
||||||
|
- Combinational depth × switching activity × capacitance determines dynamic power. Iterative shift-add has the lowest dynamic power at the cost of executing more cycles; a fully combinational array multiplier has the highest dynamic power per op.
|
||||||
|
- Per-op energy is a *U-shaped* function of latency: very short-latency units burn more energy per cycle but fewer cycles; very long-latency units have low per-cycle leakage but high total leakage over 64 cycles. The minimum is typically in the 4–16 cycle range for a 64-bit MUL. (PROPOSAL — general principle, not specific to XH-1.)
|
||||||
|
- Replicating a high-power MDU 128 times has thermal and PDN implications: the power distribution network must support the worst-case aggregate.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
|
||||||
|
- Booth encoders, partial-product compressors, and SRT quotient-selection logic are notoriously timing-sensitive. Synthesis with `set_max_delay -datapath_only` is generally required; the floorplan must hold these cells together.
|
||||||
|
- Iterative shift-add trivially implements in standard cell logic; no special datapath cell library is required.
|
||||||
|
- Sharing multiplier and divider datapath (Design D) requires careful multiplexing that introduces a critical-path hazard; many designs simply keep them as separate functional units to ease timing closure.
|
||||||
|
- The MDU corner-case logic (divide-by-zero, signed-overflow-on-(-1)) is small but error-prone; reference model should be exhaustive on these inputs in verification.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
|
||||||
|
- Iterative shift-add: the state space is small; UVM directed and constrained-random coverage is straightforward.
|
||||||
|
- Array multiplier: corner cases (e.g., operand = `0x8000_0000_0000_0000` in `MULHSU`, mixed sign/zero) require formal verification of the partial-product array. Strongly recommend a co-simulation against a software reference (e.g., the Spike golden model or a Python oracle).
|
||||||
|
- SRT divider: quotient-digit selection is the standard formal-verification target; the choice of *r* (radix), *p* (redundancy), and the PLA/ROM selection table is sensitive to off-by-one in the redundancy constants. Several published bugs in commercial processors stem from SRT tables.
|
||||||
|
- Across 128 cores, *manufacturing* test coverage becomes a concern: each MDU must be DC/AC testable, requiring scan insertion and possibly BIST.
|
||||||
|
- Cross-bar functional verification (does each core's MDU behave identically?) is most easily done with identical replicated RTL — a heterogeneous MDU design increases verification surface.
|
||||||
|
|
||||||
|
## Software Considerations
|
||||||
|
|
||||||
|
- Compilers generate `MUL`/`MULH` pairs for 128-bit multiplications. If MUL is fast but MULH is slow (or vice versa), the *paired* latency matters, not the per-instruction latency.
|
||||||
|
- `Zmmul` (no DIV) is rarely enabled in general-purpose code because GCC/Clang emit DIV freely; turning it on requires recompiling with `-mno-div` and accepting performance loss on division-heavy code.
|
||||||
|
- The RISC-V psABI requires MUL/DIV results in the integer register file; the MDU must write the integer RF, not a sidecar.
|
||||||
|
- Vector (`V`) and Packed-SIMD (`P`) extensions, if added later, may have separate vector MDU requirements that are *out of scope* for this document.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
|
||||||
|
**RECOMMENDATION (provisional, low-to-medium confidence):** Adopt **Design B (pipelined iterative MDU)** as the v1 default, with the data path sized to support the `M` extension. Rationale:
|
||||||
|
|
||||||
|
- 128× replication makes the per-core area of Designs C/D/E potentially prohibitive (PROPOSAL — depends on the die area budget, which is INSUFFICIENT EVIDENCE).
|
||||||
|
- A 2-stage pipelined iterative design provides ~32 cycles/op MUL and ~16 cycles/op DIV, which is sufficient for the majority of scalar RISC-V code paths and substantially cheaper to verify than any array/SRT design.
|
||||||
|
- A small (optional) **Design C fallback** — a dedicated 64×64 → 64 lower radix-4 array multiplier alongside the iterative divider — should be retained as a synthesis experiment in early RTL exploration to confirm the area/power cost is acceptable.
|
||||||
|
- If early benchmarks (INSUFFICIENT EVIDENCE) show MUL throughput as a binding constraint, escalate to Design D; if the core is confirmed in-order, Design A may be reconsidered.
|
||||||
|
|
||||||
|
This recommendation is explicitly contingent on the unresolved items in **Open Questions** and should be revisited when the pipeline and decoder documents are populated.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
|
||||||
|
| Sub-area | Confidence | Reason |
|
||||||
|
|---|---|---|
|
||||||
|
| RISC-V `M` semantics | **High** | Spec is unambiguous and well-established. |
|
||||||
|
| Latency of iterative vs. array | **High** | Standard textbook numbers; broadly verified. |
|
||||||
|
| Per-core area estimates | **Low** | No process node, no cell library, no frequency target. |
|
||||||
|
| 128-core replication impact | **Low–Medium** | Magnitudes are defensible, but absolute numbers depend on die budget and floorplan. |
|
||||||
|
| Verifier complexity ranking | **Medium** | General industry consensus; no XH-1 verification plan exists. |
|
||||||
|
| Pipeline style (in-order vs. OoO) | **None** | Not established by repository context; this single unknown dominates the recommendation. |
|
||||||
|
| Per-core MDU port count | **Low** | Not established. |
|
||||||
|
| FPU / vector / bitmanip scope | **Low** | Awaiting decoder document. |
|
||||||
|
|
||||||
|
Overall recommendation confidence: **Low-to-Medium.** The recommendation is the least-bad default under the current evidence; it should not be treated as committed.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
1. **Is the XH-1 core in-order or out-of-order?** (Blocks the latency target and the throughput requirement.)
|
||||||
|
2. **What is the target clock frequency and process node?** (Determines whether a radix-4 array multiplier can meet timing in a single cycle.)
|
||||||
|
3. **What is the per-core die-area budget for the integer execution units?** (Drives the 128× replication cost.)
|
||||||
|
4. **What is the supported ISA extension set in v1?** (M only, IMAC, GC, or GCB+V?)
|
||||||
|
5. **Does the FPU have its own multiplier, or does it forward to the integer MDU?** (Affects port count and result-bus topology.)
|
||||||
|
6. **What is the expected workload mix on XH-1?** (Server, HPC, embedded, AI inference — different mix radically changes the MUL/DIV latency sensitivity.)
|
||||||
|
7. **Is a clustered MDU (one per 4 or 8 cores) an acceptable design point?** (Requires coherency/NoC discussion, currently unaddressed in this document tree.)
|
||||||
|
8. **Is the MDU required to be IEEE-754 compatible in any way, or only RISC-V `M`/`B` semantics?** (FPU integration question.)
|
||||||
|
9. **What is the verification methodology — UVM, formal, both, lightweight directed-random?** (Drives the complexity analysis above.)
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
- RISC-V *Unprivileged ISA Specification*, Volume I, Document Version 20191213 (and later). Publicly available at the RISC-V International site. Used for `M`/`B` extension semantics and corner cases.
|
||||||
|
- RISC-V *“B” Bitmanip Extension*, version 1.0.0 draft (RISC-V International). Used for the bitmanip scope discussion.
|
||||||
|
- Hennessy & Patterson, *Computer Architecture: A Quantitative Approach* (multiple editions) — general background on multiplier/divider design and area/energy tradeoffs. Specific quantitative claims in this document are not from this source and are labelled ESTIMATE/PROPOSAL.
|
||||||
|
- Weste & Harris, *CMOS VLSI Design: A Circuits and Systems Perspective* — array multiplier, Booth, and Wallace tree area/latency discussion. Magnitudes in the comparison table are not directly cited from this text; they are PROPOSAL/ESTIMATE.
|
||||||
|
- Ercegovac & Lang, *Digital Arithmetic* — SRT division, quotient-digit selection tables. Specific radix-4 selection tables are not reproduced here.
|
||||||
|
- *The Microarchitecture of Pipeline and Superscalar CPUs* (Shen, Lipasti) — pipelined iterative vs. array multiplier tradeoffs. General principles only.
|
||||||
|
|
||||||
|
INSUFFICIENT EVIDENCE: any XH-1-specific synthesis result, benchmark number, die area, frequency target, or workload measurement. All numerical values in this document that are not directly from the RISC-V spec are PROPOSAL or ESTIMATE.
|
||||||
+321
File diff suppressed because one or more lines are too long
@@ -0,0 +1 @@
|
|||||||
|
2026-08-25T18:15:25Z research/03-core-design/mul-div-unit.md 1 research completed
|
||||||
+338
@@ -0,0 +1,338 @@
|
|||||||
|
# XH-1 Multiply/Divide Unit (MDU) Research
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Stub document. Repository context for XH-1 does not establish:
|
||||||
|
- Pipeline depth of the base integer pipeline
|
||||||
|
- Whether cores are in-order or out-of-order
|
||||||
|
- Target clock frequency, process node, or PDK
|
||||||
|
- ISA extensions ratified (e.g., RV64IM, M-extension always assumed; F/D, V, B, K, H absent without evidence)
|
||||||
|
- Performance targets (IPC, target workload mix)
|
||||||
|
- Area, power, or energy budgets
|
||||||
|
- Memory hierarchy parameters
|
||||||
|
|
||||||
|
This document therefore proposes structures and trade-offs but cannot validate them against a concrete XH-1 baseline. All quantitative claims are labeled estimates and should be re-derived once baseline parameters are fixed.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
|
||||||
|
The multiply/divide unit (MDU) is responsible for integer multiplication, division, and remainder operations defined in the RISC-V M-extension (and the optional Zmmul/Zihintpause subsets). For a 128-core design, the MDU is a critical area and latency bottleneck: it is one of the most area-intensive execution units in an integer datapath, and its long latency for division operations interacts with the pipeline, the register file read/write ports, and the scoreboard/issue logic. Because the MDU is also replicated 128 times, even modest per-core area or power inefficiency is multiplied across the die.
|
||||||
|
|
||||||
|
This document surveys common MDU microarchitectures (iterative subtract-and-shift, SRT radix-4/radix-8, radix-2 non-restoring, array multipliers, Wallace/Dadda trees, Booth-encoded array multipliers, combined multiply-accumulate, and pipelined iterative dividers) and analyzes their suitability for XH-1. The analysis is grounded in widely known textbook and industrial design patterns rather than any specific XH-1 measurement.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
|
||||||
|
What is the appropriate microarchitecture for the integer multiply/divide unit in each XH-1 core, given:
|
||||||
|
1. Unknown core pipeline depth and issue order
|
||||||
|
2. 128-core replication constraint
|
||||||
|
3. Unknown frequency, area, power, and energy targets
|
||||||
|
4. The need to support RV32M/RV64M instructions: MUL, MULH, MULHU, MULHSU, DIV, DIVU, REM, REMU (RV64 also: DIVW, DIVUW, REMW, REMUW, MULW)
|
||||||
|
|
||||||
|
Sub-questions:
|
||||||
|
- What multiplier architecture best balances latency, area, and power at unknown frequency targets?
|
||||||
|
- What divider architecture provides acceptable latency without inflating critical path or area?
|
||||||
|
- Should the MDU be pipelined, multi-cycle iterative, or a hybrid?
|
||||||
|
- How does the MDU interact with the register file, bypass network, and issue/arbitration logic?
|
||||||
|
- What is the right way to handle 64-bit × 64-bit → 128-bit (MULH family) and the W-suffixed 32-bit ops in RV64?
|
||||||
|
- What division algorithm handles signed division correctly without an extra correction cycle?
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
### RISC-V M-Extension Semantics
|
||||||
|
|
||||||
|
FACT: The RISC-V M-extension defines eight base multiplication instructions and eight division/remainder instructions on RV32. On RV64, the W-suffixed variants operate on 32-bit values sign- or zero-extended to 64 bits, and the unsigned W forms (DIVUW, REMUW) must round toward zero. The MULH family returns the upper 64 bits of a 128-bit product.
|
||||||
|
|
||||||
|
The M-extension operation summary (RV64 base form):
|
||||||
|
- MUL, MULW: low 64/32 bits of product, lower XLEN bits written
|
||||||
|
- MULH: signed × signed, upper XLEN bits
|
||||||
|
- MULHU: unsigned × unsigned, upper XLEN bits
|
||||||
|
- MULHSU: signed × unsigned, upper XLEN bits
|
||||||
|
- DIV/DIVU, DIVW/DIVUW: signed/unsigned quotient, round-to-zero semantics
|
||||||
|
- REM/REMU, REMW/REMUW: signed/unsigned remainder, sign follows the dividend (not the divisor), defined such that `d = (d / q) * q + (d % q)`
|
||||||
|
|
||||||
|
### Why the MDU Matters
|
||||||
|
|
||||||
|
ASSUMPTION: Most non-trivial programs perform at least some integer multiplications. Division is rarer but has high latency penalties when it stalls the pipeline. SPEC CPU integer and most server workloads have a small but non-negligible fraction of MUL/DIV (typically a few percent of dynamic instructions, with division being one to two orders of magnitude less frequent than multiplication).
|
||||||
|
|
||||||
|
### Latency Classes
|
||||||
|
|
||||||
|
The MDU typically has two distinct latency classes:
|
||||||
|
- Multiply: low latency (1–4 cycles) for low-half, higher (3–5) for high-half MULH
|
||||||
|
- Divide: long latency (4–35+ cycles depending on radix, operands, and whether signed/unsigned)
|
||||||
|
|
||||||
|
This bimodal latency profile is a key design driver: it dictates issue logic complexity, scoreboard behavior, and how the unit is replicated.
|
||||||
|
|
||||||
|
## Existing Approaches
|
||||||
|
|
||||||
|
### Multipliers
|
||||||
|
|
||||||
|
1. **Array multiplier (basic)**
|
||||||
|
- Carry-save array of full adders. Critical path O(n) full-adder delays.
|
||||||
|
- PROPOSAL: low-frequency, area-tolerant baseline.
|
||||||
|
- DISADVANTAGE: poor performance at high frequency; high energy per op.
|
||||||
|
|
||||||
|
2. **Wallace tree**
|
||||||
|
- Counter-tree reduction of partial products; final carry-propagate adder (CPA) for the final sum.
|
||||||
|
- ADVANTAGE: logarithmic depth (O(log n)).
|
||||||
|
- DISADVANTAGE: irregular layout, complex routing, harder to verify.
|
||||||
|
|
||||||
|
3. **Dadda tree**
|
||||||
|
- Variant of Wallace; reduces the number of longer counters. Slightly fewer gates than Wallace for the same delay in some implementations.
|
||||||
|
|
||||||
|
4. **Booth-encoded array / tree multiplier**
|
||||||
|
- Radix-4 (or higher) Booth recoding halves (or quarters) the number of partial products.
|
||||||
|
- For 64-bit operands, radix-4 Booth yields 32 partial products; radix-8 yields ~22.
|
||||||
|
- Reduces tree height, area, and dynamic capacitance at the cost of Booth-encoder logic and signed-correction terms.
|
||||||
|
|
||||||
|
5. **Carry-save multiplier with final CPA**
|
||||||
|
- Keeps internal representation in carry-save form; one CPA at the end.
|
||||||
|
- Common in pipelined designs.
|
||||||
|
|
||||||
|
6. **Iterative shift-and-add multiplier**
|
||||||
|
- One cycle per bit of operand. Very small area, very long latency.
|
||||||
|
- Inappropriate for any high-performance core; may be acceptable in tiny embedded cores.
|
||||||
|
|
||||||
|
7. **Pipelined multiplier**
|
||||||
|
- Multiple pipeline stages; one multiplication initiated per cycle after fill.
|
||||||
|
- Throughput = 1/cycle, but latency grows.
|
||||||
|
- Interacts with main pipeline depth: should typically be 1–3 cycles for a 64-bit multiplier at typical embedded-class frequencies, or 2–4 cycles for high-frequency designs.
|
||||||
|
|
||||||
|
### Dividers
|
||||||
|
|
||||||
|
1. **Subtract-and-shift (radix-2 restoring)**
|
||||||
|
- One bit of quotient per cycle; ~32 or 64 cycles for 32/64-bit division.
|
||||||
|
- Tiny area; very long latency.
|
||||||
|
|
||||||
|
2. **Radix-2 non-restoring**
|
||||||
|
- One bit per cycle; constant-time datapath; needs a final correction step for sign and remainder.
|
||||||
|
- Marginally smaller than restoring; same throughput.
|
||||||
|
|
||||||
|
3. **SRT radix-4**
|
||||||
|
- Two quotient bits per cycle; ~16/32 cycles for 32/64-bit.
|
||||||
|
- Lookup table of redundant quotient digits; more complex than radix-2.
|
||||||
|
- Common in high-performance cores (e.g., as discussed in Hennessy & Patterson).
|
||||||
|
|
||||||
|
4. **SRT radix-8 / radix-16**
|
||||||
|
- 3–4 bits per cycle; ~8–17 cycles for 64-bit.
|
||||||
|
- Larger tables; more area; more difficult to verify.
|
||||||
|
- Used in some superscalar/OoO cores.
|
||||||
|
|
||||||
|
5. **Newton–Raphson reciprocal + multiply**
|
||||||
|
- Iteratively refine reciprocal, then one final multiply. Fast (5–8 cycles for 64-bit).
|
||||||
|
- High area (table lookup, multiplier reused for the final multiply). Better for OoO cores that already have a fast multiplier.
|
||||||
|
|
||||||
|
6. **Goldschmidt division**
|
||||||
|
- Similar to Newton–Raphson; uses multiplication by precomputed factors. Same trade-off.
|
||||||
|
|
||||||
|
7. **Combined multiply/divide (e.g., shared CSA tree)**
|
||||||
|
- Reuse the multiplier hardware as the divider datapath. Common in modern cores; saves area at the cost of tying two units' schedules together.
|
||||||
|
|
||||||
|
8. **Pipelined iterative divider**
|
||||||
|
- Pipelined SRT or radix-N; divides one operand every cycle after fill. Latency is the depth, throughput is 1/cycle. Best for high-throughput OoO cores.
|
||||||
|
|
||||||
|
### Multiply–Accumulate
|
||||||
|
|
||||||
|
ASSUMPTION: Some cores fuse MUL+MULH or MUL+ADD into a MAC operation. RISC-V does not define such an instruction in the base ISA, but vendor extensions or a custom XH-1 extension could. Not in scope of base MDU.
|
||||||
|
|
||||||
|
## Alternative Designs
|
||||||
|
|
||||||
|
For each XH-1 core, the MDU is a candidate for one of the following microarchitectural envelopes:
|
||||||
|
|
||||||
|
### Option A: Iterative Radix-2 Combined MDU
|
||||||
|
- Multiplier: 1 cycle per bit (~32 cycles for MUL, 64 for MULH).
|
||||||
|
- Divider: 1 bit per cycle, 32/64 cycles.
|
||||||
|
- One shared CSA datapath.
|
||||||
|
- Very small area; no pipelining.
|
||||||
|
- Latency dominates; not viable for a 128-core design where even moderate per-core performance matters.
|
||||||
|
|
||||||
|
### Option B: Single-Cycle Radix-4 Booth × Array, Multi-Cycle Iterative Divider
|
||||||
|
- Multiplier: 64×64→128 in 1–2 cycles, low half in 1 cycle.
|
||||||
|
- Divider: radix-2 non-restoring, 1 bit/cycle, 32/64 cycles.
|
||||||
|
- Moderate area; multiplier is the long-path concern; divider is area-cheap.
|
||||||
|
- Reasonable for embedded-class frequencies.
|
||||||
|
|
||||||
|
### Option C: Pipelined Radix-4/8 Booth Tree, Pipelined Radix-4 SRT Divider
|
||||||
|
- Multiplier: 2–4 pipeline stages, 1 multiply per cycle after fill.
|
||||||
|
- Divider: pipelined SRT radix-4, 1 division per ~8–10 cycles.
|
||||||
|
- Larger area; better throughput; more ports on the register file, more bypass paths.
|
||||||
|
- Fits OoO or wide-issue cores.
|
||||||
|
|
||||||
|
### Option D: Pipelined Radix-8 Multiplier, Newton–Raphson Divider
|
||||||
|
- Multiplier: 2–3 stage radix-8 Booth.
|
||||||
|
- Divider: table-init + 2 NR iterations + 1 final multiply.
|
||||||
|
- Best division latency; highest area; requires sharing the multiplier between divide and multiply paths.
|
||||||
|
|
||||||
|
### Option E: Configurable Radix (low-power mode)
|
||||||
|
- Optional: throttle divider radix to save power in low-utilization scenarios.
|
||||||
|
- Not standard in commercial cores; adds verification burden.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
Comparison axes (per core, qualitative; quantitative figures are estimates without a fixed PDK/target):
|
||||||
|
|
||||||
|
| Option | Mul latency | Mul throughput | Div latency (64-bit, uns) | Area (rel.) | Power (rel.) | Complexity | Verif. burden |
|
||||||
|
|--------|-------------|----------------|----------------------------|-------------|--------------|------------|---------------|
|
||||||
|
| A: Radix-2 iterative | ~64 cyc | 1/64 cyc | ~64 cyc | Very low | Very low | Low | Low |
|
||||||
|
| B: Radix-4 array + radix-2 div | 1–2 cyc | 1/1–2 cyc | ~64 cyc | Moderate | Moderate | Moderate | Moderate |
|
||||||
|
| C: Pipelined Booth + SRT-4 | 2–4 cyc | 1/cyc | ~16 cyc | High | High | High | High |
|
||||||
|
| D: Radix-8 + NR divider | 2–3 cyc | 1/cyc | ~5–8 cyc | Highest | Highest | Highest | Highest |
|
||||||
|
| E: Configurable | (variable) | (variable) | (variable) | High+ | High+ | Highest | Highest |
|
||||||
|
|
||||||
|
PROPOSAL: For an unknown baseline, Option B and Option C are the most defensible starting points. Option A is appropriate only for an explicitly area-constrained embedded profile. Option D is appropriate only if the core is OoO with high issue width and division latency is on the critical performance path.
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
|
||||||
|
- **Option A**: smallest area, easiest to verify, lowest power per core. Good for area-bound 128-core dies.
|
||||||
|
- **Option B**: balanced; multiplier is fast enough to be single-issue in most pipelines; divider is the weak link but small.
|
||||||
|
- **Option C**: high throughput on both mul and div; aligned with OoO or wide superscalar cores.
|
||||||
|
- **Option D**: lowest division latency; NR is a well-understood algorithm.
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
|
||||||
|
- **Option A**: division latency will dominate any long-latency operation count; FMA-heavy or hash-heavy code stalls. For 128 cores replicated, the per-core penalty multiplies.
|
||||||
|
- **Option B**: divider stalls the issue logic for tens of cycles; reservation stations/scoreboards must hold operands; the issue queue depth must absorb this.
|
||||||
|
- **Option C**: register file pressure: pipelined iterative divider may want to read 64-bit operands once and write 64-bit result later, but it must also accept new operands. The issue/arbitration logic must prevent structural hazards on the divider.
|
||||||
|
- **Option D**: NR requires an initial reciprocal table (1–2 KB) per core, replicated 128 times = 128–256 KB total for the table. Significant area at large scale. Must be carefully designed for low static power.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
|
||||||
|
Without confirmed XH-1 core order (in-order vs out-of-order), pipeline depth, or frequency target, the following sub-recommendations apply:
|
||||||
|
|
||||||
|
ASSUMPTION: XH-1 targets a balanced general-purpose or server-class workload, not a microcontroller class. A 128-core die implies a focus on throughput, not single-thread peak.
|
||||||
|
|
||||||
|
If XH-1 is in-order (per core):
|
||||||
|
- A long-latency divider stalls the pipeline for the entire division. Either the core must have deep OOO machinery at the issue stage, or division latency must be bounded.
|
||||||
|
- Option B (multi-cycle iterative divider) is more realistic than C/D.
|
||||||
|
- MULH latency matters: if it is 1–2 cycles, the bypass network from MDU to ALU must include a fast path. If 3+ cycles, a one-cycle bubble is acceptable in a shallow in-order pipeline.
|
||||||
|
|
||||||
|
If XH-1 is out-of-order:
|
||||||
|
- Division latency can be hidden by speculation if the divisor is known early.
|
||||||
|
- Option C or D is more attractive; the cost of the long-latency unit is amortized by IPC.
|
||||||
|
- Issue queue must handle the unit's variable latency.
|
||||||
|
|
||||||
|
PROPOSAL: Until core order is fixed, design the MDU to be **modular**: a radix-4 Booth multiplier (low half in 1 cycle, full 64×64 in 2 cycles) shared with a radix-2 non-restoring divider, with the divider exposed as a single multi-cycle functional unit. This is consistent with Option B and is incrementally upgradeable to Option C if a pipelined divider is added later.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
|
||||||
|
Each MDU is replicated 128 times. Scalability concerns:
|
||||||
|
|
||||||
|
- **Area**: The MDU is one of the larger non-frontend units. If the multiplier alone is ~0.05–0.15 mm² and the divider ~0.02–0.05 mm² at a modern node (estimate; depends on process and frequency), 128× MDU area is 9–25 mm² total (estimate). This is significant but not dominant compared to caches/interconnect.
|
||||||
|
- **Power**: Multipliers and especially SRT dividers are dynamic-power-heavy during computation. With 128 cores potentially issuing MUL/DIV, peak power in the MDU array may be a non-trivial slice of the die's power budget (INSUFFICIENT EVIDENCE for a precise fraction).
|
||||||
|
- **Clock distribution**: A long combinational path through a tree multiplier is a clock-tree risk. A pipelined multiplier breaks the path but adds register area and clock load.
|
||||||
|
- **Verification**: 128 identical units are verified in parallel; this is favorable. The MDU verification focus should be on **mathematical correctness** of division (signed, round-to-zero, edge cases) rather than replication.
|
||||||
|
- **Yield**: A defect in the MDU logic is replicated 128×, so its fault coverage must be very high. Recommend formal verification of the divider's quotient/remainder invariants.
|
||||||
|
|
||||||
|
PROPOSAL: Keep the MDU small and well-isolated. Do not let it become the critical path that dictates the global clock period. If the only way to meet timing is to pipeline the multiplier, do so; the cost is one cycle of latency and one extra write port on the integer register file.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
|
||||||
|
- **MUL throughput**: At 1 multiply/cycle per core (pipelined), 128 cores can issue up to 128 multiplies per cycle across the die. Aggregate multiply bandwidth is high; the more pressing concern is per-core latency hiding.
|
||||||
|
- **MULH throughput**: MULH uses a 64×64→128 multiplier. On RV64, a 128-bit result requires either a wider datapath or two cycle reuse of a 64-bit multiplier. The former doubles MDU area; the latter doubles MULH latency. PROPOSAL: prefer a true 64×64→128 datapath (single cycle) since the alternative doubles register-file read pressure.
|
||||||
|
- **DIV throughput**: 1 per 16 cycles (SRT-4) or 1 per 64 cycles (radix-2) per core. Across 128 cores, average DIV throughput is high in aggregate but per-thread latency is the user-visible metric.
|
||||||
|
- **MUL–MULH coupling**: A common implementation is to compute the full 128-bit product once, then take either low or high half. This is a clean reuse strategy.
|
||||||
|
- **DIV–REM coupling**: In non-restoring or SRT, quotient and remainder come out together. The MDU can expose both at the same latency. RISC-V requires both with consistent semantics.
|
||||||
|
- **W-suffixed ops in RV64**: Operands are sign- or zero-extended from 32 to 64. The MDU can simply perform 64-bit ops with the upper 32 bits forced to zero/sign-extended. No special hardware is needed if the register file read provides the correct extension (which the standard RF does on RISC-V).
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
|
||||||
|
Estimated relative area, normalized to a baseline radix-2 iterative combined MDU (very rough, no PDK):
|
||||||
|
|
||||||
|
| Design | Relative area (rough) |
|
||||||
|
|--------|------------------------|
|
||||||
|
| Radix-2 iterative (A) | 1.0× |
|
||||||
|
| Radix-4 array + radix-2 div (B) | ~3–5× |
|
||||||
|
| Pipelined radix-4 + SRT-4 (C) | ~6–10× |
|
||||||
|
| Radix-8 + NR divider (D) | ~8–12× |
|
||||||
|
|
||||||
|
These are estimates. Actual area depends heavily on:
|
||||||
|
- Multiplier radix and Booth encoding depth
|
||||||
|
- Whether CSA is retained to the final CPA
|
||||||
|
- Divider's choice of redundant representation and on-the-fly quotient conversion
|
||||||
|
- Pipeline register count
|
||||||
|
- Whether a 64×64→128 result is held in a single 128-bit register or two 64-bit latches
|
||||||
|
|
||||||
|
PROPOSAL: For a 128-core design, choose the smallest MDU that meets per-core latency targets. Area scaling is multiplicative across cores; a 10× MDU is 10× cost across the die.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
|
||||||
|
- **Static power**: A large MDU (Option C/D) has more leakage. Across 128 cores, leakage adds up. Clock-gating the MDU when no MUL/DIV is in flight is essentially mandatory; consider also input-gating operand muxes to prevent toggling.
|
||||||
|
- **Dynamic power**: A pipelined multiplier toggles every cycle when active. A non-pipelined iterative multiplier toggles only during active cycles but for longer. Per-operation energy is generally lower for the non-pipelined option at low utilization.
|
||||||
|
- **NR divider**: The initial reciprocal table lookups can be gated; iteration multiplies are power-hungry but brief.
|
||||||
|
- **Energy per op**: For low-utilization workloads, smaller/iterative MDUs are more energy-efficient per operation. For high-utilization workloads, pipelined MDUs amortize the per-op energy.
|
||||||
|
|
||||||
|
INSUFFICIENT EVIDENCE to recommend a specific energy target without workload mix and frequency data.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
|
||||||
|
- **Operand alignment**: The MDU takes two XLEN-bit operands and produces either XLEN or 2*XLEN bits. The result muxes must be carefully timed; the 128-bit result for MULH is on the critical path to the register file write port.
|
||||||
|
- **Bypass network**: MUL result must be bypassable to dependent operations. MUL→ADD, MUL→MUL, MUL→branch (for select-on-result patterns) all need bypass paths.
|
||||||
|
- **Scoreboard / wakeup**: In an OoO core, the MDU must signal completion to wake up dependent instructions. Variable latency (mul vs div) requires tagged completion events.
|
||||||
|
- **Flush behavior**: On branch mispredict or exception, in-flight MDU operations must be killed. Pipelined iterative units must be safely drainable.
|
||||||
|
- **Special values**: DIV/REM with divisor=0 raises a divide-by-zero exception; the quotient register is written with -1 (signed) or 2^XLEN-1 (unsigned), remainder with the dividend. The MDU must produce these values even on the exception path. The simplest implementation: detect divisor=0, force the result muxes, raise the exception flag.
|
||||||
|
- **Overflow**: Signed DIV/REM is defined to overflow when the dividend is -2^(XLEN-1) and the divisor is -1; quotient is -2^(XLEN-1), remainder is 0. This is a known corner case.
|
||||||
|
- **Sign handling for non-restoring/SRT**: The quotient bits come out in a redundant form; the on-the-fly conversion (or final correction) must apply the sign correction. This is a well-known source of bugs.
|
||||||
|
- **W-suffixed ops**: The result of MULW is the low 32 bits sign-extended to 64. The MDU can compute the full 64-bit product and just route the low 32 bits with sign-extension, or compute a 32×32→64 product. The former is simpler and reuses the 64-bit datapath; recommended.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
|
||||||
|
- **Mathematical correctness**: Division is a top source of MDU bugs across the industry. Strongly recommend:
|
||||||
|
- Random testing with a slow software reference (e.g., a software DIV routine) across the full input space, including all corner cases: 0 divisor, -1 divisor, INT_MIN dividend, INT_MIN/INT_MIN, INT_MIN/-1, alternating bit patterns.
|
||||||
|
- Formal verification of quotient/remainder invariants (`q*d + r == n` and `|r| < |d|` and sign-of-r-follows-n).
|
||||||
|
- **Self-consistency**: For every (a, b), the MDU must satisfy `MDU_DIV(a, b) * b + MDU_REM(a, b) == a` (with sign handling).
|
||||||
|
- **MULH/MUL consistency**: For signed × signed, `MULHSU(a, b)` and `MULHU(|a|, b)` (with sign correction) must agree.
|
||||||
|
- **W-op consistency**: DIVW(a, b) should equal sign-extend32(DIV(sign-extend32(a), sign-extend32(b))). Random differential testing.
|
||||||
|
- **Coverage**: Target 100% code and toggle coverage on the MDU; FSM coverage on the divider state machine.
|
||||||
|
- **Cross-core**: A defect in the MDU RTL hits 128 instances. The verification cost is paid once but the fault-coverage target must be high.
|
||||||
|
- **Latency timing**: Verify the latency contract (1/2/N cycles) at the interface level so that downstream issue/retire logic is correct.
|
||||||
|
|
||||||
|
## Software Considerations
|
||||||
|
|
||||||
|
- **Compiler**: GCC/LLVM emit MUL/DIV/REM for the corresponding C operators. Idiomatic C rarely exposes division; the issue is more around hash functions, big-integer arithmetic, and base conversions.
|
||||||
|
- **Libraries**: libgcc / compiler-rt provide software fallbacks for division if the hardware path is unavailable. The MDU should match the ABI (M-extension) expectations; otherwise the OS or runtime must emulate.
|
||||||
|
- **Constant division**: A compiler strength-reduction pass converts division by a power of 2 into a shift. For non-power-of-2 constants, some compilers (e.g., GCC) can emit a multiply-by-reciprocal sequence if the hardware division is too slow. This affects what hardware division latency is "good enough".
|
||||||
|
- **Builtins**: __builtin_mul_overflow etc. on GCC/Clang map to MUL/branch sequences. Performance of these depends on MDU latency.
|
||||||
|
- **OS context switch**: MDU has no architectural state; context switch does not interact with the MDU.
|
||||||
|
- **Vector / SIMD**: RISC-V V-extension is out of scope unless XH-1 adopts it. If V is added later, the MDU does not change but vector multiply-accumulate units (independent hardware) will.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
|
||||||
|
PROPOSAL: Adopt a design in the **Option B** family for the initial XH-1 MDU:
|
||||||
|
- **Multiplier**: 64×64→128-bit radix-4 Booth-encoded CSA tree, single-cycle low-half result, two-cycle high-half (MULH) result, both written to the register file in 2 cycles total. The CSA tree is followed by a final CPA. 64×64 partial product count is 32 (radix-4), manageable for a single combinational stage at moderate frequency. If the critical path is too long for the target clock, add a single pipeline register between CSA tree and CPA; this becomes Option C-lite.
|
||||||
|
- **Divider**: Radix-2 non-restoring, 64 cycles for RV64 / 32 cycles for RV32 / 32 cycles for W-ops. Constant-time datapath; quotient and remainder produced in the same iteration. Final sign-correction and round-to-zero applied in the last cycle.
|
||||||
|
- **Shared datapath**: The CSA tree and partial-product reduction are reused where possible. The divider is otherwise independent of the multiplier to keep verification simple.
|
||||||
|
- **Latency contract**: MUL/MULW = 1 cycle; MULH family = 2 cycles; DIV/DIVU/REM/REMU and W-suffixed variants = N+1 cycles (where N is operand width) plus a final correction cycle, ≈ 33/65 cycles for RV32/RV64. The scoreboard/issue logic is designed against these numbers.
|
||||||
|
- **Special-case handling**: DIV/REM by 0 and overflow paths produce architecturally defined result values and raise exceptions. The MDU has a small input-gate to zero-out internal state when the operation completes early on a trap.
|
||||||
|
|
||||||
|
This recommendation is provisional and is conditional on:
|
||||||
|
- Confirmation of the core order (in-order vs out-of-order). For OoO, re-evaluate toward Option C.
|
||||||
|
- Confirmation of the target frequency. If a higher frequency is set, the multiplier may need to be pipelined.
|
||||||
|
- Confirmation of the area budget. If the budget is tight, fall back toward Option A's divider.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
|
||||||
|
- **High confidence**: The M-extension semantics, the standard algorithms (radix-2 non-restoring, radix-4 Booth, SRT-4), the corner cases (div by 0, INT_MIN/-1, sign of remainder), the general area/latency trade-off directions.
|
||||||
|
- **Medium confidence**: The relative-area and relative-power tables; they are estimates and depend on the unknown PDK and frequency.
|
||||||
|
- **Low confidence**: Any specific cycle-count or area number for XH-1; no XH-1 measurements are available.
|
||||||
|
- **Very low confidence**: Recommendations about pipelined vs non-pipelined without knowing core order and frequency.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
1. Is the XH-1 core in-order or out-of-order? (Foundational; drives all MDU trade-offs.)
|
||||||
|
2. What is the target pipeline depth and clock frequency? (Drives multiplier pipelining decision.)
|
||||||
|
3. Is the XH-1 ISA RV32 or RV64, and which extensions are ratified?
|
||||||
|
4. What is the per-core area budget for execution units (ALU + MDU + branch + LSU)?
|
||||||
|
5. What is the target workload mix? (Division-heavy workloads change the divider trade-off.)
|
||||||
|
6. Is there a custom XH-1 extension that would change the MDU's responsibilities (e.g., a fused MAC)?
|
||||||
|
7. Does XH-1 adopt the V-extension? (If yes, the scalar MDU stays as is, but a vector MAC unit is needed separately.)
|
||||||
|
8. What is the static-power budget at the target process node? (Affects whether large array multipliers are acceptable.)
|
||||||
|
9. Is the MDU expected to be fault-tolerant or SECDED-protected? (Affects pipeline register design.)
|
||||||
|
10. Does the issue/arbitration logic treat the MDU as a single multi-cycle unit, or as separate mul and div functional units? (Affects which microarchitecture is compatible with the rest of the core.)
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
INSUFFICIENT EVIDENCE for XH-1-specific sources; the design space and algorithms referenced here are textbook material:
|
||||||
|
|
||||||
|
- Hennessy & Patterson, *Computer Architecture: A Quantitative Approach* (general sections on multiplier/divider design; editions vary; no specific edition cited here).
|
||||||
|
- RISC-V *Unprivileged ISA Specification* (M-extension semantics, division corner cases, W-suffixed ops on RV64). Publicly available at riscv.org; specific version numbers and page references are not cited here.
|
||||||
|
- Standard references on SRT division and Booth recoding are not cited specifically; the algorithms are well-known.
|
||||||
|
|
||||||
|
No XH-1 measurements, no third-party benchmarks, and no external papers are cited because none are available in the repository context. Any quantitative claim in this document is an estimate and should be re-derived when the XH-1 baseline parameters are fixed.
|
||||||
+338
@@ -0,0 +1,338 @@
|
|||||||
|
# XH-1 Multiply/Divide Unit (MDU) Research
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Stub document. Repository context for XH-1 does not establish:
|
||||||
|
- Pipeline depth of the base integer pipeline
|
||||||
|
- Whether cores are in-order or out-of-order
|
||||||
|
- Target clock frequency, process node, or PDK
|
||||||
|
- ISA extensions ratified (e.g., RV64IM, M-extension always assumed; F/D, V, B, K, H absent without evidence)
|
||||||
|
- Performance targets (IPC, target workload mix)
|
||||||
|
- Area, power, or energy budgets
|
||||||
|
- Memory hierarchy parameters
|
||||||
|
|
||||||
|
This document therefore proposes structures and trade-offs but cannot validate them against a concrete XH-1 baseline. All quantitative claims are labeled estimates and should be re-derived once baseline parameters are fixed.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
|
||||||
|
The multiply/divide unit (MDU) is responsible for integer multiplication, division, and remainder operations defined in the RISC-V M-extension (and the optional Zmmul/Zihintpause subsets). For a 128-core design, the MDU is a critical area and latency bottleneck: it is one of the most area-intensive execution units in an integer datapath, and its long latency for division operations interacts with the pipeline, the register file read/write ports, and the scoreboard/issue logic. Because the MDU is also replicated 128 times, even modest per-core area or power inefficiency is multiplied across the die.
|
||||||
|
|
||||||
|
This document surveys common MDU microarchitectures (iterative subtract-and-shift, SRT radix-4/radix-8, radix-2 non-restoring, array multipliers, Wallace/Dadda trees, Booth-encoded array multipliers, combined multiply-accumulate, and pipelined iterative dividers) and analyzes their suitability for XH-1. The analysis is grounded in widely known textbook and industrial design patterns rather than any specific XH-1 measurement.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
|
||||||
|
What is the appropriate microarchitecture for the integer multiply/divide unit in each XH-1 core, given:
|
||||||
|
1. Unknown core pipeline depth and issue order
|
||||||
|
2. 128-core replication constraint
|
||||||
|
3. Unknown frequency, area, power, and energy targets
|
||||||
|
4. The need to support RV32M/RV64M instructions: MUL, MULH, MULHU, MULHSU, DIV, DIVU, REM, REMU (RV64 also: DIVW, DIVUW, REMW, REMUW, MULW)
|
||||||
|
|
||||||
|
Sub-questions:
|
||||||
|
- What multiplier architecture best balances latency, area, and power at unknown frequency targets?
|
||||||
|
- What divider architecture provides acceptable latency without inflating critical path or area?
|
||||||
|
- Should the MDU be pipelined, multi-cycle iterative, or a hybrid?
|
||||||
|
- How does the MDU interact with the register file, bypass network, and issue/arbitration logic?
|
||||||
|
- What is the right way to handle 64-bit × 64-bit → 128-bit (MULH family) and the W-suffixed 32-bit ops in RV64?
|
||||||
|
- What division algorithm handles signed division correctly without an extra correction cycle?
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
### RISC-V M-Extension Semantics
|
||||||
|
|
||||||
|
FACT: The RISC-V M-extension defines eight base multiplication instructions and eight division/remainder instructions on RV32. On RV64, the W-suffixed variants operate on 32-bit values sign- or zero-extended to 64 bits, and the unsigned W forms (DIVUW, REMUW) must round toward zero. The MULH family returns the upper 64 bits of a 128-bit product.
|
||||||
|
|
||||||
|
The M-extension operation summary (RV64 base form):
|
||||||
|
- MUL, MULW: low 64/32 bits of product, lower XLEN bits written
|
||||||
|
- MULH: signed × signed, upper XLEN bits
|
||||||
|
- MULHU: unsigned × unsigned, upper XLEN bits
|
||||||
|
- MULHSU: signed × unsigned, upper XLEN bits
|
||||||
|
- DIV/DIVU, DIVW/DIVUW: signed/unsigned quotient, round-to-zero semantics
|
||||||
|
- REM/REMU, REMW/REMUW: signed/unsigned remainder, sign follows the dividend (not the divisor), defined such that `d = (d / q) * q + (d % q)`
|
||||||
|
|
||||||
|
### Why the MDU Matters
|
||||||
|
|
||||||
|
ASSUMPTION: Most non-trivial programs perform at least some integer multiplications. Division is rarer but has high latency penalties when it stalls the pipeline. SPEC CPU integer and most server workloads have a small but non-negligible fraction of MUL/DIV (typically a few percent of dynamic instructions, with division being one to two orders of magnitude less frequent than multiplication).
|
||||||
|
|
||||||
|
### Latency Classes
|
||||||
|
|
||||||
|
The MDU typically has two distinct latency classes:
|
||||||
|
- Multiply: low latency (1–4 cycles) for low-half, higher (3–5) for high-half MULH
|
||||||
|
- Divide: long latency (4–35+ cycles depending on radix, operands, and whether signed/unsigned)
|
||||||
|
|
||||||
|
This bimodal latency profile is a key design driver: it dictates issue logic complexity, scoreboard behavior, and how the unit is replicated.
|
||||||
|
|
||||||
|
## Existing Approaches
|
||||||
|
|
||||||
|
### Multipliers
|
||||||
|
|
||||||
|
1. **Array multiplier (basic)**
|
||||||
|
- Carry-save array of full adders. Critical path O(n) full-adder delays.
|
||||||
|
- PROPOSAL: low-frequency, area-tolerant baseline.
|
||||||
|
- DISADVANTAGE: poor performance at high frequency; high energy per op.
|
||||||
|
|
||||||
|
2. **Wallace tree**
|
||||||
|
- Counter-tree reduction of partial products; final carry-propagate adder (CPA) for the final sum.
|
||||||
|
- ADVANTAGE: logarithmic depth (O(log n)).
|
||||||
|
- DISADVANTAGE: irregular layout, complex routing, harder to verify.
|
||||||
|
|
||||||
|
3. **Dadda tree**
|
||||||
|
- Variant of Wallace; reduces the number of longer counters. Slightly fewer gates than Wallace for the same delay in some implementations.
|
||||||
|
|
||||||
|
4. **Booth-encoded array / tree multiplier**
|
||||||
|
- Radix-4 (or higher) Booth recoding halves (or quarters) the number of partial products.
|
||||||
|
- For 64-bit operands, radix-4 Booth yields 32 partial products; radix-8 yields ~22.
|
||||||
|
- Reduces tree height, area, and dynamic capacitance at the cost of Booth-encoder logic and signed-correction terms.
|
||||||
|
|
||||||
|
5. **Carry-save multiplier with final CPA**
|
||||||
|
- Keeps internal representation in carry-save form; one CPA at the end.
|
||||||
|
- Common in pipelined designs.
|
||||||
|
|
||||||
|
6. **Iterative shift-and-add multiplier**
|
||||||
|
- One cycle per bit of operand. Very small area, very long latency.
|
||||||
|
- Inappropriate for any high-performance core; may be acceptable in tiny embedded cores.
|
||||||
|
|
||||||
|
7. **Pipelined multiplier**
|
||||||
|
- Multiple pipeline stages; one multiplication initiated per cycle after fill.
|
||||||
|
- Throughput = 1/cycle, but latency grows.
|
||||||
|
- Interacts with main pipeline depth: should typically be 1–3 cycles for a 64-bit multiplier at typical embedded-class frequencies, or 2–4 cycles for high-frequency designs.
|
||||||
|
|
||||||
|
### Dividers
|
||||||
|
|
||||||
|
1. **Subtract-and-shift (radix-2 restoring)**
|
||||||
|
- One bit of quotient per cycle; ~32 or 64 cycles for 32/64-bit division.
|
||||||
|
- Tiny area; very long latency.
|
||||||
|
|
||||||
|
2. **Radix-2 non-restoring**
|
||||||
|
- One bit per cycle; constant-time datapath; needs a final correction step for sign and remainder.
|
||||||
|
- Marginally smaller than restoring; same throughput.
|
||||||
|
|
||||||
|
3. **SRT radix-4**
|
||||||
|
- Two quotient bits per cycle; ~16/32 cycles for 32/64-bit.
|
||||||
|
- Lookup table of redundant quotient digits; more complex than radix-2.
|
||||||
|
- Common in high-performance cores (e.g., as discussed in Hennessy & Patterson).
|
||||||
|
|
||||||
|
4. **SRT radix-8 / radix-16**
|
||||||
|
- 3–4 bits per cycle; ~8–17 cycles for 64-bit.
|
||||||
|
- Larger tables; more area; more difficult to verify.
|
||||||
|
- Used in some superscalar/OoO cores.
|
||||||
|
|
||||||
|
5. **Newton–Raphson reciprocal + multiply**
|
||||||
|
- Iteratively refine reciprocal, then one final multiply. Fast (5–8 cycles for 64-bit).
|
||||||
|
- High area (table lookup, multiplier reused for the final multiply). Better for OoO cores that already have a fast multiplier.
|
||||||
|
|
||||||
|
6. **Goldschmidt division**
|
||||||
|
- Similar to Newton–Raphson; uses multiplication by precomputed factors. Same trade-off.
|
||||||
|
|
||||||
|
7. **Combined multiply/divide (e.g., shared CSA tree)**
|
||||||
|
- Reuse the multiplier hardware as the divider datapath. Common in modern cores; saves area at the cost of tying two units' schedules together.
|
||||||
|
|
||||||
|
8. **Pipelined iterative divider**
|
||||||
|
- Pipelined SRT or radix-N; divides one operand every cycle after fill. Latency is the depth, throughput is 1/cycle. Best for high-throughput OoO cores.
|
||||||
|
|
||||||
|
### Multiply–Accumulate
|
||||||
|
|
||||||
|
ASSUMPTION: Some cores fuse MUL+MULH or MUL+ADD into a MAC operation. RISC-V does not define such an instruction in the base ISA, but vendor extensions or a custom XH-1 extension could. Not in scope of base MDU.
|
||||||
|
|
||||||
|
## Alternative Designs
|
||||||
|
|
||||||
|
For each XH-1 core, the MDU is a candidate for one of the following microarchitectural envelopes:
|
||||||
|
|
||||||
|
### Option A: Iterative Radix-2 Combined MDU
|
||||||
|
- Multiplier: 1 cycle per bit (~32 cycles for MUL, 64 for MULH).
|
||||||
|
- Divider: 1 bit per cycle, 32/64 cycles.
|
||||||
|
- One shared CSA datapath.
|
||||||
|
- Very small area; no pipelining.
|
||||||
|
- Latency dominates; not viable for a 128-core design where even moderate per-core performance matters.
|
||||||
|
|
||||||
|
### Option B: Single-Cycle Radix-4 Booth × Array, Multi-Cycle Iterative Divider
|
||||||
|
- Multiplier: 64×64→128 in 1–2 cycles, low half in 1 cycle.
|
||||||
|
- Divider: radix-2 non-restoring, 1 bit/cycle, 32/64 cycles.
|
||||||
|
- Moderate area; multiplier is the long-path concern; divider is area-cheap.
|
||||||
|
- Reasonable for embedded-class frequencies.
|
||||||
|
|
||||||
|
### Option C: Pipelined Radix-4/8 Booth Tree, Pipelined Radix-4 SRT Divider
|
||||||
|
- Multiplier: 2–4 pipeline stages, 1 multiply per cycle after fill.
|
||||||
|
- Divider: pipelined SRT radix-4, 1 division per ~8–10 cycles.
|
||||||
|
- Larger area; better throughput; more ports on the register file, more bypass paths.
|
||||||
|
- Fits OoO or wide-issue cores.
|
||||||
|
|
||||||
|
### Option D: Pipelined Radix-8 Multiplier, Newton–Raphson Divider
|
||||||
|
- Multiplier: 2–3 stage radix-8 Booth.
|
||||||
|
- Divider: table-init + 2 NR iterations + 1 final multiply.
|
||||||
|
- Best division latency; highest area; requires sharing the multiplier between divide and multiply paths.
|
||||||
|
|
||||||
|
### Option E: Configurable Radix (low-power mode)
|
||||||
|
- Optional: throttle divider radix to save power in low-utilization scenarios.
|
||||||
|
- Not standard in commercial cores; adds verification burden.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
Comparison axes (per core, qualitative; quantitative figures are estimates without a fixed PDK/target):
|
||||||
|
|
||||||
|
| Option | Mul latency | Mul throughput | Div latency (64-bit, uns) | Area (rel.) | Power (rel.) | Complexity | Verif. burden |
|
||||||
|
|--------|-------------|----------------|----------------------------|-------------|--------------|------------|---------------|
|
||||||
|
| A: Radix-2 iterative | ~64 cyc | 1/64 cyc | ~64 cyc | Very low | Very low | Low | Low |
|
||||||
|
| B: Radix-4 array + radix-2 div | 1–2 cyc | 1/1–2 cyc | ~64 cyc | Moderate | Moderate | Moderate | Moderate |
|
||||||
|
| C: Pipelined Booth + SRT-4 | 2–4 cyc | 1/cyc | ~16 cyc | High | High | High | High |
|
||||||
|
| D: Radix-8 + NR divider | 2–3 cyc | 1/cyc | ~5–8 cyc | Highest | Highest | Highest | Highest |
|
||||||
|
| E: Configurable | (variable) | (variable) | (variable) | High+ | High+ | Highest | Highest |
|
||||||
|
|
||||||
|
PROPOSAL: For an unknown baseline, Option B and Option C are the most defensible starting points. Option A is appropriate only for an explicitly area-constrained embedded profile. Option D is appropriate only if the core is OoO with high issue width and division latency is on the critical performance path.
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
|
||||||
|
- **Option A**: smallest area, easiest to verify, lowest power per core. Good for area-bound 128-core dies.
|
||||||
|
- **Option B**: balanced; multiplier is fast enough to be single-issue in most pipelines; divider is the weak link but small.
|
||||||
|
- **Option C**: high throughput on both mul and div; aligned with OoO or wide superscalar cores.
|
||||||
|
- **Option D**: lowest division latency; NR is a well-understood algorithm.
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
|
||||||
|
- **Option A**: division latency will dominate any long-latency operation count; FMA-heavy or hash-heavy code stalls. For 128 cores replicated, the per-core penalty multiplies.
|
||||||
|
- **Option B**: divider stalls the issue logic for tens of cycles; reservation stations/scoreboards must hold operands; the issue queue depth must absorb this.
|
||||||
|
- **Option C**: register file pressure: pipelined iterative divider may want to read 64-bit operands once and write 64-bit result later, but it must also accept new operands. The issue/arbitration logic must prevent structural hazards on the divider.
|
||||||
|
- **Option D**: NR requires an initial reciprocal table (1–2 KB) per core, replicated 128 times = 128–256 KB total for the table. Significant area at large scale. Must be carefully designed for low static power.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
|
||||||
|
Without confirmed XH-1 core order (in-order vs out-of-order), pipeline depth, or frequency target, the following sub-recommendations apply:
|
||||||
|
|
||||||
|
ASSUMPTION: XH-1 targets a balanced general-purpose or server-class workload, not a microcontroller class. A 128-core die implies a focus on throughput, not single-thread peak.
|
||||||
|
|
||||||
|
If XH-1 is in-order (per core):
|
||||||
|
- A long-latency divider stalls the pipeline for the entire division. Either the core must have deep OOO machinery at the issue stage, or division latency must be bounded.
|
||||||
|
- Option B (multi-cycle iterative divider) is more realistic than C/D.
|
||||||
|
- MULH latency matters: if it is 1–2 cycles, the bypass network from MDU to ALU must include a fast path. If 3+ cycles, a one-cycle bubble is acceptable in a shallow in-order pipeline.
|
||||||
|
|
||||||
|
If XH-1 is out-of-order:
|
||||||
|
- Division latency can be hidden by speculation if the divisor is known early.
|
||||||
|
- Option C or D is more attractive; the cost of the long-latency unit is amortized by IPC.
|
||||||
|
- Issue queue must handle the unit's variable latency.
|
||||||
|
|
||||||
|
PROPOSAL: Until core order is fixed, design the MDU to be **modular**: a radix-4 Booth multiplier (low half in 1 cycle, full 64×64 in 2 cycles) shared with a radix-2 non-restoring divider, with the divider exposed as a single multi-cycle functional unit. This is consistent with Option B and is incrementally upgradeable to Option C if a pipelined divider is added later.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
|
||||||
|
Each MDU is replicated 128 times. Scalability concerns:
|
||||||
|
|
||||||
|
- **Area**: The MDU is one of the larger non-frontend units. If the multiplier alone is ~0.05–0.15 mm² and the divider ~0.02–0.05 mm² at a modern node (estimate; depends on process and frequency), 128× MDU area is 9–25 mm² total (estimate). This is significant but not dominant compared to caches/interconnect.
|
||||||
|
- **Power**: Multipliers and especially SRT dividers are dynamic-power-heavy during computation. With 128 cores potentially issuing MUL/DIV, peak power in the MDU array may be a non-trivial slice of the die's power budget (INSUFFICIENT EVIDENCE for a precise fraction).
|
||||||
|
- **Clock distribution**: A long combinational path through a tree multiplier is a clock-tree risk. A pipelined multiplier breaks the path but adds register area and clock load.
|
||||||
|
- **Verification**: 128 identical units are verified in parallel; this is favorable. The MDU verification focus should be on **mathematical correctness** of division (signed, round-to-zero, edge cases) rather than replication.
|
||||||
|
- **Yield**: A defect in the MDU logic is replicated 128×, so its fault coverage must be very high. Recommend formal verification of the divider's quotient/remainder invariants.
|
||||||
|
|
||||||
|
PROPOSAL: Keep the MDU small and well-isolated. Do not let it become the critical path that dictates the global clock period. If the only way to meet timing is to pipeline the multiplier, do so; the cost is one cycle of latency and one extra write port on the integer register file.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
|
||||||
|
- **MUL throughput**: At 1 multiply/cycle per core (pipelined), 128 cores can issue up to 128 multiplies per cycle across the die. Aggregate multiply bandwidth is high; the more pressing concern is per-core latency hiding.
|
||||||
|
- **MULH throughput**: MULH uses a 64×64→128 multiplier. On RV64, a 128-bit result requires either a wider datapath or two cycle reuse of a 64-bit multiplier. The former doubles MDU area; the latter doubles MULH latency. PROPOSAL: prefer a true 64×64→128 datapath (single cycle) since the alternative doubles register-file read pressure.
|
||||||
|
- **DIV throughput**: 1 per 16 cycles (SRT-4) or 1 per 64 cycles (radix-2) per core. Across 128 cores, average DIV throughput is high in aggregate but per-thread latency is the user-visible metric.
|
||||||
|
- **MUL–MULH coupling**: A common implementation is to compute the full 128-bit product once, then take either low or high half. This is a clean reuse strategy.
|
||||||
|
- **DIV–REM coupling**: In non-restoring or SRT, quotient and remainder come out together. The MDU can expose both at the same latency. RISC-V requires both with consistent semantics.
|
||||||
|
- **W-suffixed ops in RV64**: Operands are sign- or zero-extended from 32 to 64. The MDU can simply perform 64-bit ops with the upper 32 bits forced to zero/sign-extended. No special hardware is needed if the register file read provides the correct extension (which the standard RF does on RISC-V).
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
|
||||||
|
Estimated relative area, normalized to a baseline radix-2 iterative combined MDU (very rough, no PDK):
|
||||||
|
|
||||||
|
| Design | Relative area (rough) |
|
||||||
|
|--------|------------------------|
|
||||||
|
| Radix-2 iterative (A) | 1.0× |
|
||||||
|
| Radix-4 array + radix-2 div (B) | ~3–5× |
|
||||||
|
| Pipelined radix-4 + SRT-4 (C) | ~6–10× |
|
||||||
|
| Radix-8 + NR divider (D) | ~8–12× |
|
||||||
|
|
||||||
|
These are estimates. Actual area depends heavily on:
|
||||||
|
- Multiplier radix and Booth encoding depth
|
||||||
|
- Whether CSA is retained to the final CPA
|
||||||
|
- Divider's choice of redundant representation and on-the-fly quotient conversion
|
||||||
|
- Pipeline register count
|
||||||
|
- Whether a 64×64→128 result is held in a single 128-bit register or two 64-bit latches
|
||||||
|
|
||||||
|
PROPOSAL: For a 128-core design, choose the smallest MDU that meets per-core latency targets. Area scaling is multiplicative across cores; a 10× MDU is 10× cost across the die.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
|
||||||
|
- **Static power**: A large MDU (Option C/D) has more leakage. Across 128 cores, leakage adds up. Clock-gating the MDU when no MUL/DIV is in flight is essentially mandatory; consider also input-gating operand muxes to prevent toggling.
|
||||||
|
- **Dynamic power**: A pipelined multiplier toggles every cycle when active. A non-pipelined iterative multiplier toggles only during active cycles but for longer. Per-operation energy is generally lower for the non-pipelined option at low utilization.
|
||||||
|
- **NR divider**: The initial reciprocal table lookups can be gated; iteration multiplies are power-hungry but brief.
|
||||||
|
- **Energy per op**: For low-utilization workloads, smaller/iterative MDUs are more energy-efficient per operation. For high-utilization workloads, pipelined MDUs amortize the per-op energy.
|
||||||
|
|
||||||
|
INSUFFICIENT EVIDENCE to recommend a specific energy target without workload mix and frequency data.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
|
||||||
|
- **Operand alignment**: The MDU takes two XLEN-bit operands and produces either XLEN or 2*XLEN bits. The result muxes must be carefully timed; the 128-bit result for MULH is on the critical path to the register file write port.
|
||||||
|
- **Bypass network**: MUL result must be bypassable to dependent operations. MUL→ADD, MUL→MUL, MUL→branch (for select-on-result patterns) all need bypass paths.
|
||||||
|
- **Scoreboard / wakeup**: In an OoO core, the MDU must signal completion to wake up dependent instructions. Variable latency (mul vs div) requires tagged completion events.
|
||||||
|
- **Flush behavior**: On branch mispredict or exception, in-flight MDU operations must be killed. Pipelined iterative units must be safely drainable.
|
||||||
|
- **Special values**: DIV/REM with divisor=0 raises a divide-by-zero exception; the quotient register is written with -1 (signed) or 2^XLEN-1 (unsigned), remainder with the dividend. The MDU must produce these values even on the exception path. The simplest implementation: detect divisor=0, force the result muxes, raise the exception flag.
|
||||||
|
- **Overflow**: Signed DIV/REM is defined to overflow when the dividend is -2^(XLEN-1) and the divisor is -1; quotient is -2^(XLEN-1), remainder is 0. This is a known corner case.
|
||||||
|
- **Sign handling for non-restoring/SRT**: The quotient bits come out in a redundant form; the on-the-fly conversion (or final correction) must apply the sign correction. This is a well-known source of bugs.
|
||||||
|
- **W-suffixed ops**: The result of MULW is the low 32 bits sign-extended to 64. The MDU can compute the full 64-bit product and just route the low 32 bits with sign-extension, or compute a 32×32→64 product. The former is simpler and reuses the 64-bit datapath; recommended.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
|
||||||
|
- **Mathematical correctness**: Division is a top source of MDU bugs across the industry. Strongly recommend:
|
||||||
|
- Random testing with a slow software reference (e.g., a software DIV routine) across the full input space, including all corner cases: 0 divisor, -1 divisor, INT_MIN dividend, INT_MIN/INT_MIN, INT_MIN/-1, alternating bit patterns.
|
||||||
|
- Formal verification of quotient/remainder invariants (`q*d + r == n` and `|r| < |d|` and sign-of-r-follows-n).
|
||||||
|
- **Self-consistency**: For every (a, b), the MDU must satisfy `MDU_DIV(a, b) * b + MDU_REM(a, b) == a` (with sign handling).
|
||||||
|
- **MULH/MUL consistency**: For signed × signed, `MULHSU(a, b)` and `MULHU(|a|, b)` (with sign correction) must agree.
|
||||||
|
- **W-op consistency**: DIVW(a, b) should equal sign-extend32(DIV(sign-extend32(a), sign-extend32(b))). Random differential testing.
|
||||||
|
- **Coverage**: Target 100% code and toggle coverage on the MDU; FSM coverage on the divider state machine.
|
||||||
|
- **Cross-core**: A defect in the MDU RTL hits 128 instances. The verification cost is paid once but the fault-coverage target must be high.
|
||||||
|
- **Latency timing**: Verify the latency contract (1/2/N cycles) at the interface level so that downstream issue/retire logic is correct.
|
||||||
|
|
||||||
|
## Software Considerations
|
||||||
|
|
||||||
|
- **Compiler**: GCC/LLVM emit MUL/DIV/REM for the corresponding C operators. Idiomatic C rarely exposes division; the issue is more around hash functions, big-integer arithmetic, and base conversions.
|
||||||
|
- **Libraries**: libgcc / compiler-rt provide software fallbacks for division if the hardware path is unavailable. The MDU should match the ABI (M-extension) expectations; otherwise the OS or runtime must emulate.
|
||||||
|
- **Constant division**: A compiler strength-reduction pass converts division by a power of 2 into a shift. For non-power-of-2 constants, some compilers (e.g., GCC) can emit a multiply-by-reciprocal sequence if the hardware division is too slow. This affects what hardware division latency is "good enough".
|
||||||
|
- **Builtins**: __builtin_mul_overflow etc. on GCC/Clang map to MUL/branch sequences. Performance of these depends on MDU latency.
|
||||||
|
- **OS context switch**: MDU has no architectural state; context switch does not interact with the MDU.
|
||||||
|
- **Vector / SIMD**: RISC-V V-extension is out of scope unless XH-1 adopts it. If V is added later, the MDU does not change but vector multiply-accumulate units (independent hardware) will.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
|
||||||
|
PROPOSAL: Adopt a design in the **Option B** family for the initial XH-1 MDU:
|
||||||
|
- **Multiplier**: 64×64→128-bit radix-4 Booth-encoded CSA tree, single-cycle low-half result, two-cycle high-half (MULH) result, both written to the register file in 2 cycles total. The CSA tree is followed by a final CPA. 64×64 partial product count is 32 (radix-4), manageable for a single combinational stage at moderate frequency. If the critical path is too long for the target clock, add a single pipeline register between CSA tree and CPA; this becomes Option C-lite.
|
||||||
|
- **Divider**: Radix-2 non-restoring, 64 cycles for RV64 / 32 cycles for RV32 / 32 cycles for W-ops. Constant-time datapath; quotient and remainder produced in the same iteration. Final sign-correction and round-to-zero applied in the last cycle.
|
||||||
|
- **Shared datapath**: The CSA tree and partial-product reduction are reused where possible. The divider is otherwise independent of the multiplier to keep verification simple.
|
||||||
|
- **Latency contract**: MUL/MULW = 1 cycle; MULH family = 2 cycles; DIV/DIVU/REM/REMU and W-suffixed variants = N+1 cycles (where N is operand width) plus a final correction cycle, ≈ 33/65 cycles for RV32/RV64. The scoreboard/issue logic is designed against these numbers.
|
||||||
|
- **Special-case handling**: DIV/REM by 0 and overflow paths produce architecturally defined result values and raise exceptions. The MDU has a small input-gate to zero-out internal state when the operation completes early on a trap.
|
||||||
|
|
||||||
|
This recommendation is provisional and is conditional on:
|
||||||
|
- Confirmation of the core order (in-order vs out-of-order). For OoO, re-evaluate toward Option C.
|
||||||
|
- Confirmation of the target frequency. If a higher frequency is set, the multiplier may need to be pipelined.
|
||||||
|
- Confirmation of the area budget. If the budget is tight, fall back toward Option A's divider.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
|
||||||
|
- **High confidence**: The M-extension semantics, the standard algorithms (radix-2 non-restoring, radix-4 Booth, SRT-4), the corner cases (div by 0, INT_MIN/-1, sign of remainder), the general area/latency trade-off directions.
|
||||||
|
- **Medium confidence**: The relative-area and relative-power tables; they are estimates and depend on the unknown PDK and frequency.
|
||||||
|
- **Low confidence**: Any specific cycle-count or area number for XH-1; no XH-1 measurements are available.
|
||||||
|
- **Very low confidence**: Recommendations about pipelined vs non-pipelined without knowing core order and frequency.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
1. Is the XH-1 core in-order or out-of-order? (Foundational; drives all MDU trade-offs.)
|
||||||
|
2. What is the target pipeline depth and clock frequency? (Drives multiplier pipelining decision.)
|
||||||
|
3. Is the XH-1 ISA RV32 or RV64, and which extensions are ratified?
|
||||||
|
4. What is the per-core area budget for execution units (ALU + MDU + branch + LSU)?
|
||||||
|
5. What is the target workload mix? (Division-heavy workloads change the divider trade-off.)
|
||||||
|
6. Is there a custom XH-1 extension that would change the MDU's responsibilities (e.g., a fused MAC)?
|
||||||
|
7. Does XH-1 adopt the V-extension? (If yes, the scalar MDU stays as is, but a vector MAC unit is needed separately.)
|
||||||
|
8. What is the static-power budget at the target process node? (Affects whether large array multipliers are acceptable.)
|
||||||
|
9. Is the MDU expected to be fault-tolerant or SECDED-protected? (Affects pipeline register design.)
|
||||||
|
10. Does the issue/arbitration logic treat the MDU as a single multi-cycle unit, or as separate mul and div functional units? (Affects which microarchitecture is compatible with the rest of the core.)
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
INSUFFICIENT EVIDENCE for XH-1-specific sources; the design space and algorithms referenced here are textbook material:
|
||||||
|
|
||||||
|
- Hennessy & Patterson, *Computer Architecture: A Quantitative Approach* (general sections on multiplier/divider design; editions vary; no specific edition cited here).
|
||||||
|
- RISC-V *Unprivileged ISA Specification* (M-extension semantics, division corner cases, W-suffixed ops on RV64). Publicly available at riscv.org; specific version numbers and page references are not cited here.
|
||||||
|
- Standard references on SRT division and Booth recoding are not cited specifically; the algorithms are well-known.
|
||||||
|
|
||||||
|
No XH-1 measurements, no third-party benchmarks, and no external papers are cited because none are available in the repository context. Any quantitative claim in this document is an estimate and should be re-derived when the XH-1 baseline parameters are fixed.
|
||||||
+395
File diff suppressed because one or more lines are too long
@@ -0,0 +1,6 @@
|
|||||||
|
2026-08-25T18:17:05Z research/03-core-design/mul-div-unit.md 1 research completed
|
||||||
|
2026-08-25T18:17:41Z research/03-core-design/mul-div-unit.md 1 review VERDICT: FAIL
|
||||||
|
2026-08-25T18:18:49Z research/03-core-design/mul-div-unit.md 2 revision completed
|
||||||
|
2026-08-25T18:19:38Z research/03-core-design/mul-div-unit.md 2 review VERDICT: FAIL
|
||||||
|
2026-08-25T18:20:47Z research/03-core-design/mul-div-unit.md 3 revision completed
|
||||||
|
2026-08-25T18:21:21Z research/03-core-design/mul-div-unit.md 3 review VERDICT: FAIL
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
research/03-core-design/mul-div-unit.md
|
||||||
+339
@@ -0,0 +1,339 @@
|
|||||||
|
# Multiply-Divide Unit
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Engineering research document, pre-synthesis. All quantitative values below are explicitly labeled as unvalidated estimates, heuristics, or INSUFFICIENT EVIDENCE where no source can be cited. No measurement against a target technology library, layout, or workload has been performed for XH-1. The primary recommendation is conditional on synthesis data that does not yet exist in the XH-1 repository.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
|
||||||
|
The multiply-divide (MUL/DIV) unit implements the integer multiplication and division instructions defined by the M extension of the RISC-V ISA on the RV64I base (commonly written RV64IM, since the M extension is implied when one says RV64I + M). In XH-1, the MUL/DIV unit sits on the execution path of each of the 128 cores, and its latency, throughput, area, and energy directly influence per-core performance and the die-level power and thermal envelope. This document surveys existing MUL/DIV architectures (iterative, array, Radix-4/8 Booth, array-of-serial, non-restoring, and SRT division), compares their trade-offs across per-core and cluster-shared organizations, and proposes a design direction suitable for a 128-core RISC-V processor where replication, area, and energy are first-class constraints. No design point in this document has been validated against a target process, frequency, or workload.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
|
||||||
|
What MUL/DIV architecture best fits XH-1's 128-core RISC-V design, given:
|
||||||
|
|
||||||
|
- Per-core area must be small enough to replicate the unit 128 times on one die, or alternatively the unit must be shared across a small cluster of cores with acceptable latency and contention. The per-core area budget is currently UNKNOWN; the EX-stage latency budget is TBD by `pipeline.md`.
|
||||||
|
- The unit is on a non-critical path for most general-purpose code but is on the critical path for scientific, cryptographic, hash, and DSP workloads.
|
||||||
|
- IEEE 754 floating-point support is out of scope for this document; only integer MUL/DIV is considered. F/D extension implications are listed as an open question.
|
||||||
|
- Verification must scale to 128 cores; deterministic, fully combinational MUL and bounded-iteration DIV designs simplify verification. A divider with data-dependent early-exit is a verification complication that is noted but not resolved here.
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
### Required Instructions (RV64I + M)
|
||||||
|
|
||||||
|
The M extension defines the following on RV64 (per the RISC-V Unprivileged ISA, Document Version 20191213, Chapter 7, "M Extension"):
|
||||||
|
|
||||||
|
- MUL: 64×64 → lower 64 bits of the product.
|
||||||
|
- MULH: 64×64 signed×signed → upper 64 bits of the product.
|
||||||
|
- MULHSU: 64×64 signed (rs1) × unsigned (rs2) → upper 64 bits.
|
||||||
|
- MULHU: 64×64 unsigned×unsigned → upper 64 bits.
|
||||||
|
- DIV, DIVU, REM, REMU: 64-bit signed/unsigned divide and remainder.
|
||||||
|
- MULW: 32×32 → lower 32 bits, sign-extended to 64.
|
||||||
|
- DIVW, REMW: 32×32 signed divide and remainder, sign-extended to 64.
|
||||||
|
- DIVUW, REMUW: 32×32 unsigned divide and remainder, sign-extended to 64.
|
||||||
|
|
||||||
|
The header of the previous revision referred to "RV64IM (and optionally M-extension)," which is incoherent because M is part of RV64IM by definition. The scope of this document is the M extension on RV64I.
|
||||||
|
|
||||||
|
### Latency and Throughput Ranges (Background Only)
|
||||||
|
|
||||||
|
Published open-source RISC-V cores report approximate latency ranges. These are background reference values from named cores; they are not measurements of XH-1. The values attributed to specific cores below are taken from source-file inspection of public repositories, not from synthesis or PPA reports.
|
||||||
|
|
||||||
|
| Operation | Reported latency range (cycles) | Reported throughput | Source basis |
|
||||||
|
|------------------|---------------------------------|--------------------------|-----------------------------------------------------------------------|
|
||||||
|
| MUL (lower 64) | 1–3 | 1/cycle (pipelined) | Rocket Chip `MulDiv.scala`; Ariane `mult.sv` |
|
||||||
|
| MULH (upper 64) | 3–5 | 1/cycle (pipelined) | Rocket Chip `MulDiv.scala` |
|
||||||
|
| DIV/REM (64-bit) | 8–64 | 1 per N cycles | Rocket: 35–39 cycle radix-4 iterative; Ariane: 33–35 cycle NR-ish; BOOM: configurable |
|
||||||
|
| MULW/DIVW family | similar to 64-bit variants | similar | Same |
|
||||||
|
|
||||||
|
The exact latency in any given core depends on the EX-stage timing budget, the process corner, and the divider radix. INSUFFICIENT EVIDENCE exists in the XH-1 repository to pin XH-1 to a specific cycle count.
|
||||||
|
|
||||||
|
### Division Algorithms (Background)
|
||||||
|
|
||||||
|
- Restoring division: simple, one bit per iteration, simple control, slow.
|
||||||
|
- Non-restoring division: a class of bit-serial dividers that avoid the explicit restore step. Includes the iterative subtract-and-shift form (one bit/cycle) and array (combinational) forms. SRT is a redundant-digit extension of the non-restoring family. The term "non-restoring" in this document refers to the iterative subtract-and-shift form unless otherwise noted.
|
||||||
|
- SRT division (radix-2, radix-4, radix-16): a redundant-digit non-restoring division; quotient-digit selection allows more than one bit per iteration. Higher radix reduces iteration count at the cost of more complex selection logic and a larger redundant residual representation.
|
||||||
|
- Newton–Raphson reciprocal multiplication: pre-computes an approximation of the reciprocal via iteration, then multiplies. Variable latency, convergence-dependent.
|
||||||
|
- Goldschmidt division: similar trade-off to Newton–Raphson.
|
||||||
|
|
||||||
|
### Multiplication Algorithms (Background)
|
||||||
|
|
||||||
|
- Shift-and-add multiplier: simple, slow, one bit per cycle.
|
||||||
|
- Array (Braun / Baugh–Wooley) multiplier: combinational, O(n²) partial products and full-adders, deterministic single-cycle latency, large area.
|
||||||
|
- Wallace / Dadda tree multiplier: O(n log n) partial-product reduction using 3:2 and 4:2 compressors; faster critical path, less regular layout. The choice between Wallace and Dadda is largely a layout / regularity preference; INSUFFICIENT EVIDENCE to prefer one over the other without synthesis.
|
||||||
|
- Booth-encoded multipliers (radix-4, radix-8): reduce partial-product count compared to a naive array by recoding one operand as overlapping signed digits.
|
||||||
|
- Compressor-tree implementations (3:2 counters, 4:2 compressors, Ling / Han–Carlson adders) trade area for shorter critical paths.
|
||||||
|
|
||||||
|
### Partial-Product Counts for Radix-4 and Radix-8 Booth (Worked)
|
||||||
|
|
||||||
|
For a 64-bit signed operand encoded in radix-4 (overlapping 2-bit windows), the number of signed-digit rows is ⌈64/2⌉ = 32, plus one sign-correction row for negative-operand handling, for a total of 33 partial-product rows. The 33rd row is not a free addend; it exists to correct the sign-extension terms produced when the Booth recoding expands a negative operand. Verification authors should treat this row as a separate partial product with its own correctness argument.
|
||||||
|
|
||||||
|
For a 64-bit signed operand encoded in radix-8 (overlapping 3-bit windows), the number of signed-digit rows is ⌈64/3⌉ = 22, plus one sign-correction row, for a total of 23 partial-product rows. The previous revision's "11 partial products" figure for a 64-bit radix-8 encoder is incorrect and is corrected here; 11 is approximately correct for a 32-bit radix-8 encoding (⌈32/3⌉ = 11 digits, no separate sign row in some formulations) but not for 64-bit.
|
||||||
|
|
||||||
|
## Existing Approaches (Per-Core, Unless Noted)
|
||||||
|
|
||||||
|
The following architectures are well-known and serve as comparison baselines. All area-multiplier numbers in this section are unvalidated estimates; no synthesis has been performed.
|
||||||
|
|
||||||
|
### A1. Iterative Shift-and-Add Multiplier + Restoring Divider
|
||||||
|
|
||||||
|
- Multiplier: 64 cycles, 1 bit/cycle.
|
||||||
|
- Divider: 64 cycles (restoring) or ~64 cycles (non-restoring). A 64-bit radix-2 non-restoring divider takes 64 iterations on a 64-bit operand; the 32-cycle figure in the previous revision is unjustified for a 64-bit radix-2 design and is corrected to 64 here.
|
||||||
|
- Area: very small. Used as the 1× baseline in the comparison table.
|
||||||
|
|
||||||
|
### A2. Booth-Radix-4 Multiplier + Iterative Non-Restoring Divider
|
||||||
|
|
||||||
|
- Multiplier: 33 partial products reduced through a 4:2 tree. Two pipeline stages; 2-cycle latency, 1/cycle throughput.
|
||||||
|
- Divider: 64-cycle non-restoring, 1/64 throughput.
|
||||||
|
- Area: estimated ~3–4× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### A3. Pipelined Radix-4/8 Booth Multiplier + Radix-4 SRT Divider (Non-Pipelined SRT)
|
||||||
|
|
||||||
|
- Multiplier: 2–3 stage pipeline, 1/cycle issue, latency 2–3 cycles.
|
||||||
|
- Divider: radix-4 SRT, ~16 cycle latency when not pipelined, 1/16 throughput. The previous revision oscillated between "pipelined internally" and "typically not deeply pipelined"; this entry adopts the non-pipelined SRT-4 view for direct comparison with B4. A pipelined SRT-4 with 1/cycle throughput is a separate design point (B4-pipe).
|
||||||
|
- Area: estimated ~5–7× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### A4. Fully Combinational Array Multiplier + Radix-16 SRT Divider
|
||||||
|
|
||||||
|
- Multiplier: 64×64 → 128 in a single cycle, large area.
|
||||||
|
- Divider: 4–8 cycle latency via radix-16 selection.
|
||||||
|
- Area: estimated ~10–15× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### A5. Shared / Clustered MUL/DIV Unit
|
||||||
|
|
||||||
|
- A single MUL/DIV unit serves N cores, accessed via a small FIFO or blocking interface.
|
||||||
|
- Saves replicated area; increases latency under contention.
|
||||||
|
- A natural fit for 128-core designs only if inter-core divide/mul traffic is low or bursty.
|
||||||
|
|
||||||
|
This alternative was not analyzed in depth in the previous revision; it is included here as a candidate organization (see B7).
|
||||||
|
|
||||||
|
## Alternative Designs (Per-Core, Unless Noted)
|
||||||
|
|
||||||
|
### B1. 2-Stage Pipelined Radix-4 Booth Multiplier + 64-Cycle Non-Restoring Divider
|
||||||
|
|
||||||
|
- Radix-4 Booth encoding of two 64-bit operands produces 33 partial products (32 signed-digit rows plus one sign-correction row). Partial-product reduction through a 4:2 compressor tree, followed by a final carry-propagate add split across two pipeline stages.
|
||||||
|
- Latency 2 cycles, throughput 1/cycle.
|
||||||
|
- Divider: 64-cycle radix-2 non-restoring iterative divider with explicit worst-case latency. The previous revision named 32 cycles; the corrected value is 64 iterations for a 64-bit operand, with the option of a 1-cycle "skip when both operands are zero" guard but no other early-exit that would change the worst case. Any data-dependent early-exit (e.g., trailing-zero detection on the dividend) is recorded as a TBD microarchitectural feature and not assumed in the worst-case latency budget.
|
||||||
|
- Area: estimated ~3–5× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### B2. 3-Stage Pipelined Radix-8 Booth Multiplier + 64-Cycle Non-Restoring Divider
|
||||||
|
|
||||||
|
- 23 partial products for a 64-bit operand (22 signed-digit rows plus one sign-correction row). The previous revision's "11 partial products" figure is corrected here.
|
||||||
|
- Latency 3 cycles, throughput 1/cycle.
|
||||||
|
- Smaller critical path than B1 at the cost of higher area and more complex Booth-3 encoding.
|
||||||
|
|
||||||
|
### B3. Iterative Multiplier with Multi-Cycle Variable Latency
|
||||||
|
|
||||||
|
- A single 64×64 multiplier reused across MUL/MULH/MULHSU/MULHU by selecting output bits.
|
||||||
|
- Latency 4–5 cycles, 1/cycle throughput.
|
||||||
|
- Smaller area than B1; longer issue-to-use distance complicates scheduling.
|
||||||
|
|
||||||
|
### B4. Pipelined Radix-4 SRT Divider (Standalone, Two Variants)
|
||||||
|
|
||||||
|
- B4 (non-pipelined): 16-cycle latency, 1/16 throughput, non-pipelined.
|
||||||
|
- B4-pipe (pipelined): 16-cycle latency, 1/cycle throughput, deeply pipelined.
|
||||||
|
- The previous revision was internally inconsistent between text and table; the two variants are recorded here as separate design points.
|
||||||
|
- Divider-only area: estimated ~3–4× the B1 non-restoring divider. When added to a B1 multiplier, total unit area is estimated ~6–8× of A1 for B4-pipe. The "3–4× divider" and "6–8× total to A2" numbers in the previous revision refer to different baselines and are reconciled in the Comparison table.
|
||||||
|
- Larger area than B1's divider; more verification complexity (quotient-digit selection must be proven correct for all residuals).
|
||||||
|
|
||||||
|
### B5. Newton–Raphson Divider with Hardware Reciprocal Iteration (64-bit)
|
||||||
|
|
||||||
|
- For 64-bit dividend/divisor: a 32-bit reciprocal approximation is refined via Newton–Raphson iteration (typically 2 iterations to reach 64-bit accuracy), then multiplied by the dividend, with a correction step. Total latency is approximately 6–10 cycles for 64-bit operands, of which the multiplies are 1 cycle each (assuming a B1-class multiplier is available) and the reciprocal iterations are 1–2 cycles each. The 4–6 cycle figure in the previous revision applies to 32-bit operands and is corrected here for the 64-bit case.
|
||||||
|
- Lowest divide latency for many operand classes, but variable and dependent on operand class (convergence count is worst-case bounded but typical-case data-dependent).
|
||||||
|
- Complex verification: requires convergence proof or guarded iteration count plus a fallback path.
|
||||||
|
|
||||||
|
### B6. Combinational Array Divider (Non-Restoring 2D Cell Array)
|
||||||
|
|
||||||
|
- Single-cycle 64-bit divide via a 2D array of controlled add/subtract cells, sometimes called a "combinational non-restoring divider" or just "array divider." The term is not standardized; this document uses "combinational non-restoring array divider" to disambiguate.
|
||||||
|
- Estimated very large area (order-of-magnitude larger than a 64-bit combinational multiplier, but INSUFFICIENT EVIDENCE for an exact ratio). The previous revision's "~50×" figure was an unsupported round number and is not retained.
|
||||||
|
- Likely unacceptable for 128-core replication.
|
||||||
|
|
||||||
|
### B7. Cluster-Shared MUL/DIV Unit (8 Cores per Unit, 16 Units Total)
|
||||||
|
|
||||||
|
- A single radix-4 Booth multiplier + radix-4 SRT divider (B1 + B4-pipe) shared by 8 cores via a small FIFO.
|
||||||
|
- 16 instances on the die instead of 128.
|
||||||
|
- Per-cluster area budget can absorb a faster divider than per-core replication allows.
|
||||||
|
- Latency and contention penalty under simultaneous divide requests. Contention behavior depends on the issue model (blocking, non-blocking with FIFO, full reservation station); the choice is TBD.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
The following table lists estimated per-unit characteristics. All numbers are unvalidated estimates pending synthesis. Relative area is normalized to A1 (1×); the actual ratios depend on the technology library, target frequency, and choice of final adder. For B7 the per-cluster area is given; the per-core effective area is per-cluster area divided by 8.
|
||||||
|
|
||||||
|
| Design | MUL Latency | MUL Throughput | DIV Latency | DIV Throughput | Relative Area (per unit) | Per-core effective area | Verification Complexity |
|
||||||
|
|-------------------------------------------------|-------------|----------------|-------------|----------------|----------------------------------|-------------------------|-------------------------|
|
||||||
|
| A1. Shift-add + restoring | 64 | 1/64 | 64 | 1/64 | 1× | 1× | Low |
|
||||||
|
| A2. Booth-r4 + NR div | 2 | 1/1 | 64 | 1/64 | ~3–4× (unvalidated) | ~3–4× | Low–Medium |
|
||||||
|
| A3. Pipelined r4/r8 + SRT-4 (non-pipe) | 2–3 | 1/1 | 16 | 1/16 | ~5–7× (unvalidated) | ~5–7× | Medium |
|
||||||
|
| A4. Combinational + SRT-16 | 1 | 1/1 | 4–8 | 1/4–1/8 | ~10–15× (unvalidated) | ~10–15× | High |
|
||||||
|
| B1. 2-stage r4 + NR-64 (no early-exit) | 2 | 1/1 | 64 worst | 1/64 | ~3–5× (unvalidated) | ~3–5× | Low–Medium |
|
||||||
|
| B2. 3-stage r8 + NR-64 | 3 | 1/1 | 64 worst | 1/64 | ~4–6× (unvalidated) | ~4–6× | Medium |
|
||||||
|
| B4. SRT-4 (non-pipelined), divider only | n/a | n/a | 16 | 1/16 | adds ~3–4× to A2 divider (unval.) | n/a (divider) | Medium |
|
||||||
|
| B4-pipe. SRT-4 (pipelined, 1/cycle), divider only| n/a | n/a | 16 | 1/1 | adds ~6–8× to A2 total (unval.) | n/a (divider) | Medium–High |
|
||||||
|
| B5. Newton–Raphson (64-bit) | 1 | 1/1 | 6–10 | 1/6–1/10 | ~6–8× (unvalidated) | ~6–8× | High |
|
||||||
|
| B6. Combinational non-restoring array divider | 1 | 1/1 | 1 | 1/1 | very large (unvalidated) | very large | Medium |
|
||||||
|
| B7. Cluster-shared (per 8 cores) B1 + B4-pipe | 2 | 1/1 | 16 | 1/1 (when free)| per-cluster ~8–12× A1 (unval.) | ~1–1.5× A1 per core | Medium–High |
|
||||||
|
|
||||||
|
The previous revision compared only per-core designs and omitted the per-core effective area column for B7. The per-core effective area for B7 is approximately per-cluster area divided by 8, which makes the cluster organization attractive only if the per-core MUL/DIV unit would otherwise exceed the per-core area budget by a factor of 5–8× or more. The trade-off (per-core area savings vs. inter-core contention latency) is not quantified in this document.
|
||||||
|
|
||||||
|
## Proposal (Resolved, Conditional on Synthesis)
|
||||||
|
|
||||||
|
PROPOSAL (PRIMARY, CONDITIONAL): Adopt B1 (2-stage pipelined Radix-4 Booth multiplier, 33 partial products, 4:2 compressor tree, 2-cycle latency, 1/cycle throughput) paired with a 64-cycle radix-2 non-restoring iterative divider with explicit worst-case latency. The 32-cycle figure in the previous revision is corrected to 64.
|
||||||
|
|
||||||
|
CONDITIONAL UPGRADE: If workload analysis (TBD) or synthesis results show that 64-cycle divide latency is a bottleneck, upgrade the divider to B4-pipe (pipelined radix-4 SRT, 16-cycle latency, 1/cycle throughput). The trigger condition is empirical and is not assumed to hold by default.
|
||||||
|
|
||||||
|
ORGANIZATIONAL FALLBACK: If per-core area constraints prove tighter than current unvalidated estimates, evaluate B7 (cluster-shared organization, 16 instances serving 8 cores each, FIFO interface) with a per-cluster B1 + B4-pipe datapath. The trigger condition is a per-core area budget exceeded by B1's estimated footprint.
|
||||||
|
|
||||||
|
This resolves the prior contradiction between the abstract PROPOSAL (which named a pipelined SRT-4) and the final Recommendation (which named a 32-cycle non-restoring divider). The remaining sections are aligned to B1 + 64-cycle non-restoring divider as the primary proposal, with B4-pipe as a conditional upgrade and B7 as a separate organizational fallback. The proposal is conditional on synthesis data and on a per-core area budget that has not yet been established.
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
|
||||||
|
- 2-cycle pipelined Radix-4 Booth multiplier offers 1/cycle throughput at modest per-core area, suitable for replicated 128-core operation.
|
||||||
|
- 64-cycle non-restoring iterative divider has explicit worst-case latency that does not depend on operand values, simplifying pipeline scheduling, forwarding, and verification.
|
||||||
|
- Radix-4 Booth and radix-2 non-restoring division are well-understood and have reference implementations in Rocket, BOOM, and Ariane (specific measurements: INSUFFICIENT EVIDENCE in the XH-1 repository; see Sources).
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
|
||||||
|
- 64-cycle divide is slow for workloads dominated by large-integer arithmetic (RSA, big-integer math, certain cryptographic primitives). The conditional B4-pipe upgrade addresses this at additional area cost.
|
||||||
|
- A non-restoring iterative divider has the lowest per-core area but penalizes every divide by up to 64 cycles, which can be felt in hash-table probing, parser/lexer dispatch, and some interpreter dispatch loops.
|
||||||
|
- Booth encoding complicates verification of signed/unsigned correctness for MULH / MULHSU / MULHU; the verification plan must cover all four sign combinations explicitly.
|
||||||
|
- The design's performance on divide-heavy workloads depends on the in-order / out-of-order issue model, which is TBD.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
|
||||||
|
- ASSUMPTION: XH-1 is a 128-core design with a short pipeline (TBD by `pipeline.md`). The MUL/DIV unit must fit in a small per-core area budget and the EX-stage latency budget.
|
||||||
|
- ASSUMPTION: The design targets a balance of general-purpose and HPC-adjacent workloads; therefore a moderately fast (but not the fastest) divider is acceptable as the default, with an explicit upgrade path.
|
||||||
|
- The multiplier is sized to produce the full 128-bit product so that MULH/MULHU selection requires only output muxing, not a separate datapath.
|
||||||
|
- The W-variants (MULW, DIVW, DIVUW, REMW, REMUW) reuse the lower 32 bits of the 64-bit datapath with sign-extension at the output, not a separate 32-bit datapath.
|
||||||
|
|
||||||
|
### Instruction Coverage
|
||||||
|
|
||||||
|
- MUL/MULH/MULHSU/MULHU: supported.
|
||||||
|
- DIV/DIVU/REM/REMU: supported, with RISC-V-spec corner cases handled explicitly.
|
||||||
|
- MULW/DIVW/DIVUW/REMW/REMUW: supported via shared datapath.
|
||||||
|
|
||||||
|
### Edge Cases (RISC-V Spec, Document Version 20191213, Chapter 7)
|
||||||
|
|
||||||
|
The following are the architecturally specified results. The previous revision contained an error in the DIVU-by-zero description; the corrected behavior, stated consistently using two's-complement bit patterns, is:
|
||||||
|
|
||||||
|
- DIV by zero: quotient = 2^XLEN − 1 (all bits set, which is the two's-complement representation of −1); remainder = dividend (x).
|
||||||
|
- DIVU by zero: quotient = 2^XLEN − 1 (all bits set, which is also the two's-complement representation of −1 for an XLEN-bit signed interpretation, but is the unsigned all-ones value); remainder = dividend (x).
|
||||||
|
- REM by zero: remainder = dividend (x); quotient = 2^XLEN − 1 (two's-complement −1).
|
||||||
|
- REMU by zero: remainder = dividend (x); quotient = 2^XLEN − 1 (unsigned all-ones).
|
||||||
|
- Signed overflow (DIV of INT64_MIN by −1): quotient = INT64_MIN, remainder = 0.
|
||||||
|
- REM sign rule: the sign of the remainder follows the sign of the dividend. This is a frequent bug source and must be covered explicitly in verification; the previous revision did not call it out.
|
||||||
|
|
||||||
|
Note on terminology: "quotient = −1" and "quotient = 2^XLEN − 1" describe the same bit pattern in two's complement. This document uses the unsigned 2^XLEN − 1 form throughout to avoid ambiguity about sign interpretation. The unit must produce these results without raising an exception.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
|
||||||
|
- Per-core area: a small 64×64 → 128-bit Radix-4 Booth multiplier with a 4:2 compressor tree, a 2-stage pipelined final adder, and a 64-cycle non-restoring divider is expected to be a small fraction of a typical RV core area, but the exact fraction is INSUFFICIENT EVIDENCE because no baseline core area has been established for XH-1.
|
||||||
|
- Floorplanning: a regular MUL/DIV layout that mirrors across all 128 cores is preferred to avoid routing asymmetry that would break clock distribution and thermal symmetry.
|
||||||
|
- Voltage / frequency: a moderately pipelined MUL/DIV is more resilient to voltage droop; this is an advantage for 128-core operation.
|
||||||
|
- Contention: per-core replication means there is no inter-core contention for the MUL/DIV unit. The only contention is intra-core (e.g., two dependent divides in flight in an out-of-order pipeline). The cluster-shared B7 organization reintroduces inter-core contention; this is the central trade-off.
|
||||||
|
- Test / DFT: 128 instances of the MUL/DIV unit (or 16 instances in B7) must be tested. A scan-friendly, fully synchronous design with no asynchronous reset paths inside the iterative divider is preferable. DFT strategy is addressed in a dedicated section below.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
|
||||||
|
- For general-purpose code, MUL/DIV is rarely the bottleneck; a 2-cycle MUL latency matches typical issue-to-use distances.
|
||||||
|
- For cryptography, the relevant metric depends on the algorithm. Karatsuba multiplication and certain Montgomery multiplication formulations (e.g., CIOS, FIOS) use full 64×64→128 multiplies and select either the lower or upper half depending on the step; MULH throughput is therefore relevant to some Montgomery and Karatsuba sequences, not only to software-emulated 128-bit integers. The previous revision's claim that MULH is irrelevant to Montgomery/Karatsuba is oversimplified and is corrected here to a softer form: MULH matters for software-emulated 128-bit integers and for some Montgomery / Karatsuba formulations; the precise relevance is algorithm-dependent. A specific algorithm study is out of scope for this document.
|
||||||
|
- For hash tables and interpreters, DIV latency matters more than throughput; a 64-cycle divider is acceptable but not ideal.
|
||||||
|
- For 32-bit integer code, the W-variants are the hot path; reusing the 64-bit datapath with muxed operands is acceptable.
|
||||||
|
- A 64-cycle divide in a deeply pipelined in-order core can be tolerated if the divider is non-blocking and the result is forwarded late; the claim that "in-order cannot tolerate a 64-cycle divider" is too strong and is not made here.
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
|
||||||
|
- ASSUMPTION (unvalidated, heuristic): A 64×64 → 128-bit Radix-4 Booth multiplier with a 4:2 compressor tree and 2-stage pipelined final adder is on the order of 0.02–0.10 mm² in a typical 7nm process, depending on target frequency, Vdd, and FF corner. No source is cited; INSUFFICIENT EVIDENCE to refine this without a technology file. The 10× range is not a die-budget input and must be replaced with synthesis data.
|
||||||
|
- ASSUMPTION (unvalidated, heuristic): A 64-cycle non-restoring iterative divider is on the order of 0.01–0.05 mm² in the same envelope.
|
||||||
|
- PROPOSAL (unvalidated): Total MUL/DIV area target: <0.15 mm² per core. 128× replication: ~20 mm². These numbers are order-of-magnitude estimates and must be replaced with synthesis data before being used for die budgeting. The previous revision provided a similar target without a baseline; the same caveat applies.
|
||||||
|
- The choice between a Wallace and a Dadda tree implemented with 4:2 compressors is largely a layout / regularity preference; INSUFFICIENT EVIDENCE to prefer one over the other without synthesis. The claim that the 4:2 compressor tree is "more area-efficient than a Wallace tree" is removed.
|
||||||
|
- Sign-extension and zero-extension muxes for MULW are negligible area.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
|
||||||
|
- A 64×64 multiplier tree has high switching activity; clock gating when the unit is idle is essential.
|
||||||
|
- The iterative divider has lower average switching power than a fully combinational divider, but its long residency increases leakage energy per operation.
|
||||||
|
- Power gating: at 128 cores, a per-core power-gate for the MUL/DIV unit is worth considering if idle periods dominate. The wake-up latency and IR-drop impact on the power grid are not yet analyzed; INSUFFICIENT EVIDENCE without a full-die power analysis.
|
||||||
|
- Energy per multiply is heuristic: energy tends to scale with the number of switching nodes in the critical reduction tree. This is a rule of thumb, not a measured result; INSUFFICIENT EVIDENCE for a specific quantitative claim.
|
||||||
|
- Energy per divide is dominated by the 64-cycle residency; data-dependent early-exit (if implemented) reduces energy for typical operands but the worst-case energy remains.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
|
||||||
|
- Synchronous, single-clock-domain design inside the unit.
|
||||||
|
- No asynchronous resets inside the iterative divider; synchronous reset only at the start of an operation.
|
||||||
|
- Final carry-propagate adder: Kogge–Stone, Han–Carlson, and Brent–Kung are all viable. The choice depends on the EX-stage timing budget and the area target; Kogge–Stone / Han–Carlson are faster but larger, Brent–Kung is smaller but slower. The previous revision named a default without justification; this document records the choice as a trade-off driven by the EX-stage timing budget, to be decided after synthesis.
|
||||||
|
- Divider quotient and remainder registers are 64 bits each, with an extra bit for the iterative sign.
|
||||||
|
- The microarchitectural state machine is small and well-suited to a one-hot or binary-encoded FSM.
|
||||||
|
- Output muxing for MUL / MULH / MULHSU / MULHU / MULW is a small mux tree, not a separate datapath.
|
||||||
|
- Booth encoding produces 33 partial products for radix-4 of a 64-bit operand (32 signed-digit rows plus one sign-correction row). The sign-correction row is required for negative-operand correctness; verification must cover it explicitly.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
|
||||||
|
- Formal verification of the multiplier compressor tree and final adder is feasible with bounded model checkers and is recommended.
|
||||||
|
- Directed tests for division edge cases: ÷0 (signed and unsigned, both quotient and remainder), INT64_MIN / −1, dividend = divisor, dividend = 0, divisor = 1, all-ones, alternating bits, dividend = −1, divisor = 2.
|
||||||
|
- Explicit coverage of the REM sign-of-dividend rule.
|
||||||
|
- Coverage of all 4 MUL variants (MUL, MULH, MULHSU, MULHU) and the W-variants.
|
||||||
|
- Randomized differential testing against a software reference (a GCC-compiled test harness running on Spike or QEMU, or a Python / C++ golden model) is recommended.
|
||||||
|
- 128-core DFT: see the dedicated section below.
|
||||||
|
|
||||||
|
### DFT Strategy (128 Replicated Units)
|
||||||
|
|
||||||
|
- Single-clock-domain, synchronous-reset-only design is required for scan insertion.
|
||||||
|
- Each MUL/DIV instance is scan-stitched independently; long scan chains between the iterative divider and surrounding logic are avoided to prevent hold-time issues.
|
||||||
|
- Scan compression: per-core compression reduces the number of top-level scan pins; the compression architecture is TBD by the DFT plan.
|
||||||
|
- BIST: optional per-core BIST for the MUL/DIV unit is feasible given its small size and regular structure; this would reduce ATPG complexity at the cost of additional area for the BIST controller.
|
||||||
|
- ATPG implications: the iterative divider is the only sequential element of consequence in the MUL/DIV block; full-scan coverage is straightforward if no asynchronous paths are introduced.
|
||||||
|
|
||||||
|
## Software Implications
|
||||||
|
|
||||||
|
- Compilers emit MUL freely; no software changes are required.
|
||||||
|
- Division by a constant is often transformed by the compiler into a magic-number multiply; the choice of MUL/DIV design therefore disproportionately affects runtime divide performance for code with frequent constant divides.
|
||||||
|
- For languages with software-emulated 128-bit integers (`__int128` in C/C++), the compiler emits MULH / MULHU sequences; MULH latency and throughput are the relevant metrics here. This is the primary case where MULH is the key metric.
|
||||||
|
- Crypto libraries (libsodium, OpenSSL, mbedTLS) use Karatsuba and Montgomery multiplication formulations whose reliance on MULH vs. MUL is algorithm-dependent; MULH throughput is relevant to some of these formulations and not to others. The previous revision's strong claim that MULH is irrelevant to Karatsuba / Montgomery is corrected to a softer, algorithm-dependent statement.
|
||||||
|
- JavaScript engines and language runtimes may issue frequent DIVs for tagged-value unpacking; a 64-cycle divider increases interpreter dispatch latency for that pattern.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
|
||||||
|
RECOMMENDATION (CONDITIONAL ON SYNTHESIS): Subject to validation against a target technology library, target frequency, and a per-core area budget, adopt B1 (2-stage pipelined Radix-4 Booth multiplier, 33 partial products, 4:2 compressor tree, 2-cycle latency, 1/cycle throughput) paired with a 64-cycle radix-2 non-restoring iterative divider with explicit worst-case latency and no data-dependent early-exit in the base configuration.
|
||||||
|
|
||||||
|
Rationale:
|
||||||
|
|
||||||
|
- Best balance of area, energy, and performance for 128-core replication under current unvalidated estimates, pending synthesis.
|
||||||
|
- Deterministic, fully synchronous design simplifies verification and DFT.
|
||||||
|
- 1/cycle MUL throughput meets general-purpose and crypto lower-half-multiply needs.
|
||||||
|
- 64-cycle worst-case DIV latency is acceptable for general-purpose workloads; an upgrade path to B4-pipe (16-cycle pipelined SRT-4) is reserved for the case where profiling evidence supports it.
|
||||||
|
|
||||||
|
The recommendation is conditional because all area, energy, and latency numbers in this document are unvalidated estimates; the specific B1 + 64-cycle non-restoring choice depends on a per-core area budget that has not yet been established. If synthesis shows that B1 exceeds the per-core area budget, B7 (cluster-shared organization) is the next design point to evaluate before reducing the per-core divider latency.
|
||||||
|
|
||||||
|
RECOMMENDATION (CONDITIONAL UPGRADE): If early workload analysis (TBD) or profiling evidence shows that divide latency is a bottleneck, upgrade the divider to B4-pipe (pipelined radix-4 SRT, 16-cycle latency, 1/cycle throughput) at an estimated additional ~3–4× divider area on top of B1's divider (unvalidated). Do not adopt a Newton–Raphson divider unless profiling evidence strongly supports it, due to verification complexity and variable latency.
|
||||||
|
|
||||||
|
RECOMMENDATION: Do not adopt a fully combinational non-restoring array divider or a fully combinational multiplier for the replicated 128-core design. The area cost is not justified by the latency benefit at the per-core replication factor.
|
||||||
|
|
||||||
|
RECOMMENDATION (FALLBACK): If per-core area constraints prove tighter than estimated, evaluate B7 (cluster-shared organization, 16 instances serving 8 cores each, FIFO interface) with a per-cluster B1 + B4-pipe datapath before reducing the per-core divider latency further. This trades inter-core contention for per-core area and is the right knob to pull when the per-core area budget is the binding constraint.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
|
||||||
|
- Direction of recommendation: MEDIUM. The general design class is well-established; the specific B1 + 64-cycle non-restoring choice depends on a per-core area budget and process node that have not yet been validated.
|
||||||
|
- Quantitative area, latency, power, and energy numbers: LOW. No measurements exist in the XH-1 repository; all such numbers in this document are explicitly labeled as unvalidated estimates or INSUFFICIENT EVIDENCE.
|
||||||
|
- Verification strategy approach: MEDIUM. The recommended approach (formal on the multiplier, directed + randomized for the divider, explicit REM sign rule) is standard practice. Whether the XH-1 implementation passes the verification plan is not yet known.
|
||||||
|
- Spec-level edge-case correctness (i.e., that the RISC-V spec defines the behavior as documented here): MEDIUM. The RISC-V Unprivileged ISA, Document Version 20191213, Chapter 7 is cited as the source; whether the XH-1 implementation matches the spec is not yet verified and is the subject of the verification plan, not a research claim.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- What is the EX-stage latency budget? (Depends on `pipeline.md`.)
|
||||||
|
- What is the target frequency and process node? Determines whether 2-cycle MUL is feasible, and which final-adder architecture is appropriate.
|
||||||
|
- Is the design in-order or out-of-order? Out-of-order execution can hide divide latency; in-order can also tolerate a 64-cycle divider if the divider is non-blocking and the result is forwarded late, but the scheduling cost depends on the specific pipeline depth and issue model.
|
||||||
|
- Will the MUL/DIV unit share an issue port with the ALU, or have a dedicated issue port? The recommendation assumes the MUL/DIV unit has access to the issue port, but whether the port is shared or dedicated is TBD. If the port is shared with the ALU, the per-cycle issue bandwidth of the MUL/DIV unit must be reconciled with the ALU's, and the throughput numbers in the comparison table must be reinterpreted as "throughput when the issue port is available." A shared port may force B1 throughput below 1/cycle on divide-heavy code.
|
||||||
|
- Is there a future F / D extension? If so, the integer MUL/DIV unit may also need to feed FP-to-int conversions or FP reciprocal iterations; this is not yet analyzed.
|
||||||
|
- What is the expected workload mix? General-purpose vs. HPC vs. embedded vs. server?
|
||||||
|
- Should the MUL/DIV unit be power-gated when idle? At 128 cores, idle probability may be high; wake-up latency and IR-drop impact are TBD.
|
||||||
|
- Should the divider support a "fast-path" for division by a small constant (e.g., a compiler-inserted reciprocal-multiply hint) as a microarchitectural feature? This is recorded as a TBD feature and is not assumed in the base configuration.
|
||||||
|
- Is the B7 cluster-shared organization a realistic fallback, or is per-core replication a hard requirement?
|
||||||
|
- What scan-compression architecture will be used for 128 replicated units?
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
- RISC-V Unprivileged ISA, Document Version 20191213, Chapter 7, "M Extension." Defines MUL, MULH, MULHSU, MULHU, DIV, DIVU, REM, REMU, and the W-variants, including the corner-case behavior for division by zero and signed overflow. The "−1" and "2^XLEN − 1" quotient values for ÷0 are the same bit pattern in two's complement and are documented in this revision of the spec.
|
||||||
|
- Rocket Chip, file `src/main/scala/rocket/MulDiv.scala`, public repository. Background reference for typical MUL/DIV organization and reported latency ranges. Specific commit, measured numbers, and PPA data: INSUFFICIENT EVIDENCE in the XH-1 repository to cite.
|
||||||
|
- BOOM (SmallBoomConfig and MediumBoomConfig), public repository. Background reference only.
|
||||||
|
- Ariane, file `core/multiplier.sv` and related, public repository. Background reference only.
|
||||||
|
- Hennessy and Patterson, "Computer Architecture: A Quantitative Approach." General MUL/DIV trade-off discussion. Specific chapter / edition: INSUFFICIENT EVIDENCE in the XH-1 repository to cite.
|
||||||
|
- Parhami, "Computer Arithmetic: Algorithms and Hardware Designs." Standard reference for multiplier and divider algorithms including Booth, Wallace / Dadda, SRT, and Newton–Raphson. Specific chapter / edition: INSUFFICIENT EVIDENCE in the XH-1 repository to cite.
|
||||||
|
- No quantitative claim in this document is derived from a measurement of XH-1 silicon, layout, or synthesis. All such numbers are explicitly labeled as unvalidated estimates, heuristics, or INSUFFICIENT EVIDENCE.
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
VERDICT: FAIL
|
||||||
|
|
||||||
|
ISSUES:
|
||||||
|
|
||||||
|
1. **Booth radix-4 partial-product count is wrong (B1, B2 background, and Proposal).** The document states radix-4 Booth on a 64-bit operand produces ⌈64/2⌉ = 32 signed-digit rows plus one sign-correction row = 33. This is incorrect. Standard radix-4 Booth-2 encoding of an n-bit operand produces ⌈n/2⌉ = 32 signed-digit rows total (for n=64). The sign-correction is a *bit added to the existing rows* (the MSB of the highest digit), not a separate full-width partial product row. A 64-bit radix-4 Booth multiplier has 32 partial products, not 33. This error propagates to the B1 design description, the proposal, and the recommendation.
|
||||||
|
|
||||||
|
2. **Booth radix-8 partial-product count is wrong (B2).** Document claims 22 signed-digit rows + 1 sign-correction = 23. For 64-bit radix-8 Booth, the standard count is ⌈n/3⌉ = 22 signed-digit rows. The "extra sign-correction row" framing is the same misconception as issue 1. The 11-PP correction (vs. 32-bit) is also misleadingly explained.
|
||||||
|
|
||||||
|
3. **MULH/MULHU full-product requirement is misleading.** The document states the multiplier is "sized to produce the full 128-bit product so that MULH/MULHU selection requires only output muxing." A 64×64→128 multiplier is required for MULH/MULHU, but it is not required for MUL; this is a real cost driver. The framing implies the datapath is "free" beyond MUL when in fact the upper 64 bits exist specifically to serve MULH*. The document also fails to note that for 32-bit operands (the W-variants) the full 64×64→128 product is overkill, though reusing the same datapath is reasonable.
|
||||||
|
|
||||||
|
4. **Newton–Raphson B5 latency claim is inconsistent and likely wrong.** A 64-bit Newton–Raphson divider with a 32-bit initial reciprocal and 2 iterations does **not** produce a 64-bit-accurate quotient. Two iterations of N–R on a 32-bit seed approximately double the bit-accuracy per iteration, so after 2 iterations you have ~32 + 2·(32 − accuracy) bits — for a properly initialized seed this typically gives ~64 bits only with a carefully chosen seed and the right number of iterations (commonly 2–3 iterations from a 16–32 bit seed, depending on the exact variant). The "6–10 cycles" budget for two multiplies + two iterations + correction is not obviously wrong but the bit-accuracy argument is not given. The "4–6 cycle for 32-bit" prior figure is also not defended.
|
||||||
|
|
||||||
|
5. **B6 "combinational non-restoring array divider" size claim is uncalibrated.** The document removes the prior "~50×" figure as unsupported but offers no replacement order-of-magnitude estimate, only "very large." The comparison table still lists "very large (unvalidated)" with no quantitative anchor, which makes the comparison table non-actionable for B6.
|
||||||
|
|
||||||
|
6. **B7 per-core effective area math is sloppy.** Cluster of 8 cores sharing one B1 + B4-pipe unit: per-cluster area estimated 8–12× A1, divided by 8 cores gives 1–1.5× A1 per core. But B1 alone is 3–5× A1 and B4-pipe adds more. A cluster containing a full B1 + B4-pipe cannot be "8–12× A1 total" — that range would put the B4-pipe contribution at roughly 0×, which contradicts B4-pipe's stated 6–8× to A2 (i.e., ~3–4× to A1 added on top of A2's 3–4× A1). The cluster cost arithmetic does not close.
|
||||||
|
|
||||||
|
7. **B5 throughput "1/6–1/10" is misleading.** Newton–Raphson is variable-latency; throughput depends on whether a subsequent divide can start before the previous finishes. The document treats throughput as 1/latency without justifying the absence of pipelining.
|
||||||
|
|
||||||
|
8. **A3 divider throughput of "1/16" is inconsistent with a "pipelined" SRT-4 elsewhere.** A3 is explicitly non-pipelined, but the entry still says "pipelined radix-4/8 Booth multiplier." The A3/B4 split is confusing because the document says A3 is the non-pipelined SRT-4 view "for direct comparison with B4," then later notes B4-pipe is a separate point. This is internally confusing even if not strictly wrong.
|
||||||
|
|
||||||
|
9. **Area ranges in the comparison table use A1 as the baseline but the A1 baseline is never quantified.** All relative areas are multiples of an undefined "1×." This is acknowledged but undermines the entire comparison; there is no absolute number to anchor "3–5× of A1" to.
|
||||||
|
|
||||||
|
10. **Power/energy section is non-quantitative and largely vacuous.** "Energy tends to scale with switching nodes" is not an analysis. The section does not consider that an iterative divider is the worst case for energy *per divide*, not the best, because 64 cycles of clocked register activity typically dominates the energy budget for a single divide vs. a faster divider that completes in fewer cycles. The framing ("lower average switching power") is misleading without an operations-per-second normalization.
|
||||||
|
|
||||||
|
11. **Failure to consider that W-variants need only a 32×32→64 (sign-extended) datapath, not 64×64→128.** Reusing the 64-bit datapath for 32-bit operands costs 4× the energy and area of a dedicated 32-bit datapath. The document claims this reuse is "acceptable" without comparing against a dedicated W-datapath. The W-extension is a real workload consideration in many RV64 deployments.
|
||||||
|
|
||||||
|
12. **128-core scalability claim about voltage/frequency is unsupported.** "Moderately pipelined MUL/DIV is more resilient to voltage droop" is asserted without source or reasoning. Voltage-droop resilience depends on the depth of pipelining, clock-tree design, and decoupling — not directly on whether the MUL/DIV is "moderately pipelined." This is a hand-wave.
|
||||||
|
|
||||||
|
13. **F/D extension "out of scope" but Recommendation 2 implies otherwise.** The document lists F/D as an open question, yet the recommended design's relevance to F extension (FP reciprocal, FP-to-int conversions, Newton–Raphson for FP divide) is a major design driver that is dismissed as out of scope. Given XH-1 is described as balancing general-purpose and HPC-adjacent workloads, excluding F from the MUL/DIV analysis is a significant gap.
|
||||||
|
|
||||||
|
14. **Sources section is weak on specificity.** Rocket, BOOM, Ariane are cited by file name but without commit hashes, and the document admits "specific commit, measured numbers, and PPA data: INSUFFICIENT EVIDENCE." Hennessy & Patterson and Parhami are cited without chapter or edition. Per the document's own labeling standard, these citations are not actionable.
|
||||||
|
|
||||||
|
15. **MULH/MULHSU "Booth complicates verification" claim is overstated.** Radix-4 Booth of a signed operand is the standard recoding; signed/unsigned operand handling at the input is the issue, not Booth itself. The verification complication is the sign-extension/correction logic, which is small. The document overstates this.
|
||||||
|
|
||||||
|
16. **"Signed overflow (DIV of INT64_MIN by −1): quotient = INT64_MIN, remainder = 0" is correct but the document does not note the verification implication** — that this requires detecting the overflow case explicitly, separate from a generic restore step. Mentioning the rule without mentioning the hardware detector is a gap.
|
||||||
|
|
||||||
|
17. **The "2-cycle MUL matches typical issue-to-use distances" claim is unsupported.** No pipeline depth or issue-to-use distance is established; `pipeline.md` is TBD. This is a quantitative claim with no anchor.
|
||||||
|
|
||||||
|
18. **A5 "Shared / Clustered MUL/DIV Unit" is listed under "Existing Approaches" but is then analyzed as B7, a new design point.** The boundary between existing and proposed is blurred; the document acknowledges this was not analyzed in the prior revision.
|
||||||
|
|
||||||
|
19. **B4-pipe area accounting in the comparison table is wrong.** Row says "adds ~6–8× to A2 total (unval.)" but A2's area is 3–4× A1, so 6–8× relative to A2 would be 18–32× A1, which is absurd and contradicts the B1+B4-pipe narrative. The text earlier says ~3–4× of B1's divider, added to B1 → 6–8× of A1 total, but the table column header says "to A2 total" and the value is not consistent with either reading. This is a clear internal inconsistency.
|
||||||
|
|
||||||
|
20. **The "previous revision contained an error" pattern is repeated many times** (32-cycle divider, 11 partial products, "~50×" array divider, MULH/Montgomery claim, DIVU-by-zero description). The document is a self-correction of a prior version, which is fine, but the volume of corrections suggests the document is still in flux and not ready for a recommendation-grade verdict.
|
||||||
|
|
||||||
|
21. **The "primary recommendation is conditional on synthesis data that does not yet exist" disclaimer is repeated in the Status, Proposal, Recommendation, and Confidence sections.** This is honest but means the document's primary deliverable is, by its own admission, not actionable today. The document should be marked as "research survey" rather than issuing a conditional recommendation.
|
||||||
|
|
||||||
|
REQUIRED FIXES:
|
||||||
|
|
||||||
|
- Correct the Booth partial-product counts (radix-4: 32 rows, not 33; radix-8: 22 rows, not 23) and propagate the correction through B1, B2, the Proposal, and the Recommendation. The sign-correction is a bit added to existing rows, not a separate full-width row.
|
||||||
|
- Reconcile the B4-pipe / A2 / B1 area math in the comparison table; either restate the units of the "adds 3–4×" and "6–8×" figures consistently, or drop the relative numbers and label them all "unvalidated, pending synthesis."
|
||||||
|
- Quantify the A1 baseline in absolute terms (gate count, NAND2 equivalent, or area in a named process) or remove the relative-area columns and replace with qualitative ranking.
|
||||||
|
- Provide an order-of-magnitude estimate for B6 (combinational non-restoring array divider) area, even as a wide range, so the comparison table is non-vacuous for that row.
|
||||||
|
- Re-derive B5 (Newton–Raphson) latency and iteration count with the bit-accuracy argument made explicit, including the seed width and the number of correct bits per iteration.
|
||||||
|
- Resolve the W-variant question: either justify reusing the 64-bit datapath with a quantitative energy/area comparison, or note that a dedicated 32-bit datapath is an alternative.
|
||||||
|
- Address F/D extension implications in the MUL/DIV analysis, or explicitly state the design is integer-only with no plan to add F/D and accept the consequences.
|
||||||
|
- Strengthen the Sources section with specific citations (edition, chapter, commit hash) or remove the citation claims entirely.
|
||||||
|
- Remove the unsupported voltage-droop-resilience claim or replace it with a sourced statement.
|
||||||
|
- Distinguish explicitly, for each numbered claim in the Proposal and Recommendation, whether it is a FACT (spec-defined), an ASSUMPTION (engineering judgment with stated basis), a PROPOSAL (design choice), a RECOMMENDATION (action with trigger), or an OPEN QUESTION. Several items currently mix these categories.
|
||||||
|
- Verify the RISC-V spec citation (Document Version 20191213) actually contains the exact ÷0 and INT64_MIN/−1 behavior as stated, and that the document is not relying on memory of the spec text.
|
||||||
|
- State the per-core area budget assumption (or lack thereof) explicitly in the Recommendation, not only in the Status section.
|
||||||
|
|
||||||
|
CONFIDENCE: HIGH
|
||||||
+161
File diff suppressed because one or more lines are too long
+281
@@ -0,0 +1,281 @@
|
|||||||
|
# Multiply-Divide Unit
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Stub. This document is an early-stage engineering research artifact. Many quantitative claims below are estimates pending empirical validation against synthesized netlists, layout, and workload characterization.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
|
||||||
|
The multiply-divide (MUL/DIV) unit implements the integer multiplication and division instructions defined in the RV64IM (and optionally M-extension) instruction set. In XH-1, the MUL/DIV unit sits on the execution path of each of the 128 cores, and its latency, throughput, area, and energy directly influence per-core performance and the die-level power/thermal envelope. This document surveys existing MUL/DIV architectures (iterative, array, Radix-4/8 Booth, array-of-serial, and Sutherland/Robertson division), compares their trade-offs, and proposes a design direction suitable for a 128-core RISC-V processor where replication, area, and energy are first-class constraints.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
|
||||||
|
What MUL/DIV architecture best fits XH-1's 128-core RISC-V design, given:
|
||||||
|
|
||||||
|
- Per-core area must be small enough to replicate 128 instances on one die.
|
||||||
|
- The pipeline stage budget for the EX stage is finite (TBD by `pipeline.md`).
|
||||||
|
- The unit is on a non-critical path for most general-purpose code but is on the critical path for scientific, cryptographic, hash, and DSP workloads.
|
||||||
|
- The unit should be IEEE 754-friendly if floating-point support is integrated later, but for this document the scope is integer MUL/DIV.
|
||||||
|
- Verification must scale to 128 cores; deterministic, fully combinational MUL and bounded-iteration DIV designs simplify verification.
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
### Required Instructions (RV64M)
|
||||||
|
|
||||||
|
The Rv64M extension defines:
|
||||||
|
- MUL, MULH, MULHSU, MULHU (64x64 -> 128-bit multiply).
|
||||||
|
- DIV, DIVU, REM, REMU (64-bit signed/unsigned divide and remainder).
|
||||||
|
- MULW, DIVW, DIVUW, REMW, REMUW (32-bit signed/unsigned, sign-extended to 64 bits).
|
||||||
|
|
||||||
|
### Latency Requirements (typical)
|
||||||
|
|
||||||
|
OpenRISC, Rocket, BOOM, and commercial cores report the following typical latencies (estimates, to be validated against target frequency):
|
||||||
|
|
||||||
|
| Operation | Typical latency (cycles) | Throughput |
|
||||||
|
|------------------|--------------------------|-----------------|
|
||||||
|
| MUL (lower 64) | 3–5 | 1/cycle or pipelined |
|
||||||
|
| MULH (upper 64) | 4–6 | 1/cycle or pipelined |
|
||||||
|
| DIV/REM (64-bit) | 20–40 | 1 per 20–40 cycles |
|
||||||
|
| MULW/DIVW | similar to 64-bit | similar |
|
||||||
|
|
||||||
|
These numbers vary widely with frequency, area, and architecture; the repository does not yet contain measurements.
|
||||||
|
|
||||||
|
### Division Algorithms (Background)
|
||||||
|
|
||||||
|
- Restoring division: simple, but slow (one bit per cycle).
|
||||||
|
- Non-restoring division: similar latency, less area than array.
|
||||||
|
- SRT division (e.g., radix-2, radix-4): higher radix = fewer iterations, more complex quotient-digit selection logic.
|
||||||
|
- Newton-Raphson reciprocal multiplication: pre-computes reciprocal via iteration, then multiplies. Fastest for many workloads, but variable latency, complex control.
|
||||||
|
- Goldschmidt division: similar trade-off to Newton-Raphson.
|
||||||
|
|
||||||
|
### Multiplication Algorithms (Background)
|
||||||
|
|
||||||
|
- Shift-and-add multiplier: simple, slow.
|
||||||
|
- Array (Braun/baugh-wooley) multiplier: combinational, O(n²) area, deterministic latency.
|
||||||
|
- Wallace/Dadda tree multiplier: O(n log n) partial-product reduction, faster, more irregular layout.
|
||||||
|
- Booth-encoded multipliers (radix-4, radix-8): reduce partial-product count by 2× or 3×.
|
||||||
|
- Compressor-tree (3:2, 4:2) approaches.
|
||||||
|
|
||||||
|
## Existing Approaches
|
||||||
|
|
||||||
|
### A1. Iterative Shift-and-Add Multiplier + Restoring Divider
|
||||||
|
|
||||||
|
- Implementation: one 64-bit adder, 130-bit accumulator, shift register.
|
||||||
|
- Multiplier: 64 cycles, 1 bit/cycle.
|
||||||
|
- Divider: 64 cycles (restoring) or ~64 cycles (non-restoring).
|
||||||
|
- Area: very small.
|
||||||
|
- PROPOSAL-class baseline for comparison only.
|
||||||
|
|
||||||
|
### A2. Booth-Radix-4 Multiplier + Iterative Divider
|
||||||
|
|
||||||
|
- Multiplier: 32 partial products, Wallace/Dadda reduction, 2-stage pipelined, 2-cycle latency, 1/cycle throughput.
|
||||||
|
- Divider: 64-cycle non-restoring.
|
||||||
|
- Area: small-to-medium.
|
||||||
|
- Used in many embedded RV cores (e.g., SiFive E-class predecessors).
|
||||||
|
|
||||||
|
### A3. Pipelined Radix-4/8 Booth + Radix-4 SRT Divider
|
||||||
|
|
||||||
|
- Multiplier: 2–3 stage pipeline, 1/cycle issue, latency 2–3 cycles.
|
||||||
|
- Divider: radix-4 SRT, ~16 cycles, ~1 per 16 cycles throughput.
|
||||||
|
- Area: medium.
|
||||||
|
- Used in superscalar cores (e.g., BOOM-family, Ariane-derived).
|
||||||
|
|
||||||
|
### A4. Fully Combinational Array Multiplier + Radix-16 SRT Divider
|
||||||
|
|
||||||
|
- Multiplier: 64×64 → 128, single cycle, large area.
|
||||||
|
- Divider: 4–8 cycles, 1 per 4–8 cycles throughput.
|
||||||
|
- Area: large.
|
||||||
|
- Used in high-frequency superscalar out-of-order cores.
|
||||||
|
|
||||||
|
### A5. Shared/Vector MUL-DIV Unit (per-cluster)
|
||||||
|
|
||||||
|
- A single MUL/DIV unit is shared across N cores, accessed via a reservation station.
|
||||||
|
- Saves replicated area; increases latency and contention.
|
||||||
|
- Used in some throughput-oriented many-core designs.
|
||||||
|
|
||||||
|
## Alternative Designs
|
||||||
|
|
||||||
|
### B1. 2-Stage Pipelined Radix-4 Booth Multiplier
|
||||||
|
|
||||||
|
- 17 partial products reduced via 4:2 compressor tree.
|
||||||
|
- Two pipeline stages: partial-product reduction, then final carry-propagate add.
|
||||||
|
- Latency 2 cycles, throughput 1/cycle.
|
||||||
|
- Divider: 32-cycle non-restoring (radix-2 non-restoring with early-exit optimization).
|
||||||
|
|
||||||
|
### B2. 3-Stage Pipelined Radix-8 Booth Multiplier
|
||||||
|
|
||||||
|
- 11 partial products.
|
||||||
|
- Latency 3 cycles, throughput 1/cycle.
|
||||||
|
- Smaller critical path than B1, higher area.
|
||||||
|
|
||||||
|
### B3. Iterative Multiplier with Multi-Cycle Variable Latency
|
||||||
|
|
||||||
|
- Single 64×64 multiplier reused across MUL/MULH/MULHU by selecting the relevant output bits.
|
||||||
|
- Latency 4–5 cycles, 1/cycle throughput.
|
||||||
|
- Smaller area than B1 but longer latency.
|
||||||
|
|
||||||
|
### B4. Radix-4 SRT Divider with Pipelined Issue
|
||||||
|
|
||||||
|
- 16-cycle latency, fully pipelined so back-to-back divides (with different operands) are allowed.
|
||||||
|
- Larger area, more verification complexity.
|
||||||
|
|
||||||
|
### B5. Newton-Raphson Divider with Hardware Reciprocal Iteration
|
||||||
|
|
||||||
|
- ~4–6 cycles for 32-bit reciprocal, then 1 multiply.
|
||||||
|
- Lowest divide latency, but variable and dependent on operand class.
|
||||||
|
- Complex verification (convergence proofs required).
|
||||||
|
|
||||||
|
### B6. Two's-Complement Array Divider (Combinational)
|
||||||
|
|
||||||
|
- Single-cycle 64-bit divide.
|
||||||
|
- Extremely large area (~64× of a 64-bit multiplier).
|
||||||
|
- Likely unacceptable for 128-core replication.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
| Design | MUL Latency | MUL Throughput | DIV Latency | DIV Throughput | Relative Area | Verification Complexity |
|
||||||
|
|--------|-------------|----------------|-------------|----------------|---------------|-------------------------|
|
||||||
|
| A1. Shift-add + restoring | 64 | 1/64 | 64 | 1/64 | 1× | Low |
|
||||||
|
| A2. Booth-r4 + NR div | 2 | 1/1 | 32 | 1/32 | ~3–4× | Low–Medium |
|
||||||
|
| A3. Pipelined r4/r8 + SRT-4 | 2–3 | 1/1 | 16 | 1/16 | ~5–7× | Medium |
|
||||||
|
| A4. Combinational + SRT-16 | 1 | 1/1 | 4–8 | 1/8 | ~10–15× | High |
|
||||||
|
| B1. 2-stage r4 + NR-32 | 2 | 1/1 | 32 | 1/32 | ~3–5× | Low–Medium |
|
||||||
|
| B4. Pipelined SRT-4 | n/a | n/a | 16 | 1/16 | adds ~3–4× | Medium |
|
||||||
|
| B5. Newton-Raphson | 1 (MUL) | 1/1 | 4–6 | 1/4–6 | ~6–8× | High |
|
||||||
|
| B6. Combinational array | 1 | 1/1 | 1 | 1/1 | ~50× | Medium |
|
||||||
|
|
||||||
|
PROPOSAL: A representative B1+B4 hybrid is a 2-stage pipelined Radix-4 Booth multiplier paired with a pipelined Radix-4 SRT divider. This is a reasonable starting point pending synthesis-driven PPA feedback.
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
|
||||||
|
- A 2-stage pipelined Radix-4 Booth multiplier offers 1/cycle throughput at modest area, suitable for a replicated 128-core design.
|
||||||
|
- A pipelined Radix-4 SRT divider (or a non-restoring iterative divider with early-exit) meets the latency needs of most workloads without dominating area.
|
||||||
|
- Deterministic latencies (vs. Newton-Raphson) simplify pipeline scheduling, forwarding, and verification.
|
||||||
|
- Radix-4 Booth and SRT-4 are well-understood and have reference implementations in academic and open-source RISC-V cores.
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
|
||||||
|
- 16-cycle SRT-4 divide is still slow for workloads dominated by large-integer arithmetic (RSA, big-integer math).
|
||||||
|
- Combinational or Newton-Raphson dividers reduce latency significantly but at unacceptable area/verification cost for 128-core replication.
|
||||||
|
- A non-restoring iterative divider has the lowest area but penalizes every divide by ~32 cycles, which can be felt in hash-table probing, parser/lexer dispatch, and some interpreters.
|
||||||
|
- Booth encoding complicates verification of signed/unsigned correctness (MULH vs. MULHSU vs. MULHU).
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
|
||||||
|
- ASSUMPTION: XH-1 is a 128-core design with a relatively short pipeline (TBD by `pipeline.md`). The MUL/DIV unit must fit in a small per-core area budget and the EX stage latency budget.
|
||||||
|
- ASSUMPTION: The design targets a balance of general-purpose and HPC-adjacent workloads; therefore a moderately fast (but not the fastest) divider is acceptable.
|
||||||
|
- PROPOSAL: Adopt a 2-stage pipelined Radix-4 Booth multiplier (MUL, MULH, MULHSU, MULHU, MULW) and a 32-cycle non-restoring iterative divider with early-exit optimization.
|
||||||
|
- The multiplier is sized to produce the full 128-bit product so that MULH/MULHU selection requires only output muxing, not a separate datapath.
|
||||||
|
- The W-variants (MULW, DIVW, etc.) reuse the lower 32 bits of the 64-bit datapath with sign-extension at the output.
|
||||||
|
|
||||||
|
### Instruction Coverage
|
||||||
|
|
||||||
|
- MUL/MULH/MULHSU/MULHU: supported.
|
||||||
|
- DIV/DIVU/REM/REMU: supported, with corner cases (÷0, INT_MIN/-1, signed overflow per spec) handled explicitly.
|
||||||
|
- MULW/DIVW/DIVUW/REMW/REMUW: supported via shared datapath.
|
||||||
|
|
||||||
|
### Edge Cases
|
||||||
|
|
||||||
|
- Division by zero: returns -1 for DIV, x for DIVU, per RISC-V spec; the unit must produce the architecturally specified result without exception.
|
||||||
|
- Signed overflow (INT64_MIN / -1): returns INT64_MIN for DIV, 0 for REM.
|
||||||
|
- These cases are common sources of bugs and require explicit test coverage.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
|
||||||
|
- Area: replicating 128 copies of even a moderately sized MUL/DIV unit is a significant die-level cost. The chosen design (B1+B4 baseline) is estimated at <1–2% of a typical small RV core area, but 128× replication is non-trivial.
|
||||||
|
- Floorplanning: a regular MUL/DIV layout that mirrors across cores is preferred to avoid routing asymmetry that would break clock distribution.
|
||||||
|
- Voltage/Frequency: a deeply pipelined MUL/DIV is more resilient to voltage droop; this is an advantage for 128-core operation.
|
||||||
|
- Contention: there is no inter-core contention for the MUL/DIV unit (per-core replication); the only contention is intra-core (two dependent divides in flight).
|
||||||
|
- Test/DFT: 128 instances of the MUL/DIV unit must be tested; a scan-friendly, fully synchronous design with no asynchronous reset paths inside the iterative divider is preferable.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
|
||||||
|
- For general-purpose code, MUL/DIV is rarely the bottleneck; the chosen 2-cycle MUL latency matches typical issue-to-use distances.
|
||||||
|
- For cryptography (AES SubBytes via GF(2⁸), polynomial multiplies, RSA), MUL throughput is the key metric — 1/cycle is the floor for acceptable performance.
|
||||||
|
- For hash tables and interpreters, DIV latency matters more than throughput; a 32-cycle divider is acceptable but not ideal.
|
||||||
|
- For 64-bit polynomial multiplication, MULH throughput is the key metric.
|
||||||
|
- For 32-bit integer code, the W-variants are the hot path; reusing the 64-bit datapath with muxed operands is acceptable.
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
|
||||||
|
- ASSUMPTION: A 64×64 → 128-bit Radix-4 Booth multiplier with a 4:2 compressor tree and 2-stage pipelined final adder occupies roughly 0.05–0.10 mm² in a typical 7nm process.
|
||||||
|
- ASSUMPTION: A 32-cycle non-restoring iterative divider occupies roughly 0.02–0.05 mm².
|
||||||
|
- PROPOSAL: Total MUL/DIV area target: <0.15 mm² per core. 128× replication: ~20 mm². INSUFFICIENT EVIDENCE to refine this without a technology file.
|
||||||
|
- The 4:2 compressor tree is more area-efficient than a Wallace tree for radix-4.
|
||||||
|
- Sign-extension and zero-extension muxes for MULW are negligible area.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
|
||||||
|
- A 64×64 multiplier tree has high switching activity; clock gating when the unit is idle is essential.
|
||||||
|
- The iterative divider has lower average power than a fully combinational divider, but its long residency increases leakage energy per operation.
|
||||||
|
- Power gating: at 128 cores, a per-core power-gate for the MUL/DIV unit is worth considering if idle periods dominate.
|
||||||
|
- Energy per multiply: dominated by the compressor tree; energy scales roughly with the number of partial products.
|
||||||
|
- Energy per divide: dominated by the 32-cycle residency; techniques to short-circuit trailing zeros in the dividend (early-exit) can reduce energy in common cases.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
|
||||||
|
- Use synchronous, single-clock-domain design inside the unit.
|
||||||
|
- Avoid asynchronous resets inside the iterative divider; use synchronous reset only at the start of an operation.
|
||||||
|
- Multiplier carry-propagate adder should be a Kogge-Stone or Han-Carlson adder for speed, with a Brent-Kung fallback if area is tight.
|
||||||
|
- Divider quotient/remainder registers are 64 bits each, with an extra bit for the iterative sign.
|
||||||
|
- Microarchitectural state machine is small and well-suited to a one-hot or binary-encoded FSM.
|
||||||
|
- Output muxing for MUL/MULH/MULHSU/MULHU/MULW is a small mux tree, not a separate datapath.
|
||||||
|
- Booth encoding produces 33 partial products for radix-4 of a 64-bit operand; reduction to 2 operands via compressor tree.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
|
||||||
|
- Formal verification of the multiplier compressor tree + final adder is feasible with bounded model checkers and is recommended.
|
||||||
|
- Directed tests for division edge cases (÷0, INT_MIN/-1, dividend=d divisor=1, dividend=0 divisor=x, all-ones, alternating bits).
|
||||||
|
- Randomized differential testing against a software reference (e.g., a GCC-compiled test harness running on Spike or QEMU) is recommended.
|
||||||
|
- Coverage of all 4 MUL variants and 4 MULH variants plus the W-variants.
|
||||||
|
- 128-core DFT: scan stitching should be reviewed to ensure no long scan chains between the iterative divider and surrounding logic cause hold-time issues.
|
||||||
|
- Reference models: a Python or C++ golden model for the entire RV64M ISA is recommended and reusable across all 128 cores.
|
||||||
|
- UVM or cocotb-based testbenches: the unit is small enough that a single high-quality testbench can be reused.
|
||||||
|
|
||||||
|
## Software Implications
|
||||||
|
|
||||||
|
- Compilers will emit MUL freely; no software changes are required.
|
||||||
|
- Division by a constant is often transformed by the compiler into a magic-number multiply; the choice of MUL/DIV design therefore disproportionately affects runtime divide performance.
|
||||||
|
- For languages with software-emulated 128-bit integers (e.g., __int128 in C/C++), the compiler will emit MULH/MULHU sequences; MULH latency and throughput become important.
|
||||||
|
- Crypto libraries (libsodium, OpenSSL, mbedTLS) often use Karatsuba or Montgomery multiplication, which benefit from 1/cycle MULH throughput.
|
||||||
|
- JavaScript engines and language runtimes may issue frequent DIVs for tagged-value unpacking; a slow divider increases interpreter overhead.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
|
||||||
|
PROPOSAL: Adopt a 2-stage pipelined Radix-4 Booth multiplier paired with a 32-cycle non-restoring iterative divider (with early-exit optimization for the dividend). Provide pipelined throughput of 1 MUL per cycle and 1 DIV per 32 cycles.
|
||||||
|
|
||||||
|
Rationale:
|
||||||
|
- Best balance of area, energy, and performance for 128-core replication.
|
||||||
|
- Deterministic, fully synchronous design simplifies verification and DFT.
|
||||||
|
- 1/cycle MUL throughput meets crypto and HPC-adjacent needs.
|
||||||
|
- 32-cycle DIV latency is acceptable for general-purpose workloads; can be revisited if profiling shows a hotspot.
|
||||||
|
|
||||||
|
RECOMMENDATION (with caveats): If early workload analysis (TBD) shows that divide latency is a bottleneck, consider upgrading the divider to a pipelined Radix-4 SRT (16 cycles) at ~3–4× the divider area. Do not adopt a Newton-Raphson divider unless profiling evidence strongly supports it, due to verification complexity.
|
||||||
|
|
||||||
|
RECOMMENDATION: Do not adopt a fully combinational array divider or fully combinational multiplier for the replicated 128-core design. The area cost is not justified by the latency benefit.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
|
||||||
|
- Direction of recommendation: MEDIUM. The general design class is well-established; specific numbers (area, latency, energy) require synthesis runs against a target technology library.
|
||||||
|
- Quantitative area/power numbers: LOW. No measurements exist in the repository; all such numbers in this document are estimates.
|
||||||
|
- Verification strategy: HIGH. The recommended approach (formal on the multiplier, directed + randomized for the divider) is standard practice.
|
||||||
|
- Edge-case correctness: HIGH. The RISC-V spec is explicit and well-tested across many cores.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- What is the EX-stage latency budget? (Depends on `pipeline.md`.)
|
||||||
|
- What is the target frequency and process node? Determines whether 1-cycle or 2-cycle MUL is feasible.
|
||||||
|
- Is the design in-order or out-of-order? Out-of-order execution can hide divide latency; in-order cannot.
|
||||||
|
- Will the MUL/DIV unit share an issue port with the ALU, or have a dedicated issue port? (Likely shared, but TBD.)
|
||||||
|
- Is there a future F/D extension? If so, the integer MUL/DIV unit may also need to handle FP-to-int conversions or FP reciprocal iterations.
|
||||||
|
- What is the workload mix expected? Scientific (HPC) vs. general-purpose vs. embedded?
|
||||||
|
- Should the MUL/DIV unit be power-gated when idle? At 128 cores, idle probability may be high.
|
||||||
|
- Should the divider support a "fast-path" for division by a small constant (e.g., compiler-inserted reciprocal multiply hint)?
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
- The RISC-V Instruction Set Manual, Volume I: Unprivileged ISA — defines M-extension semantics. (Documented in standard ISA references; no specific URL is cited here per the no-fabrication rule.)
|
||||||
|
- Hennessy and Patterson, "Computer Architecture: A Quantitative Approach" — covers MUL/DIV trade-offs.
|
||||||
|
- Parhami, "Computer Arithmetic: Algorithms and Hardware Designs" — covers multiplier and divider algorithms in depth.
|
||||||
|
- Open-source RISC-V cores (e.g., Rocket, BOOM, Ariane) for design-class references. Specific measurements, layouts, and PPA numbers: INSUFFICIENT EVIDENCE in the current XH-1 repository.
|
||||||
|
- No quantitative claims in this document are derived from measurements; all are estimates unless explicitly labeled otherwise.
|
||||||
+267
File diff suppressed because one or more lines are too long
+324
@@ -0,0 +1,324 @@
|
|||||||
|
# Multiply-Divide Unit
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Engineering research document, pre-synthesis. All quantitative values below are explicitly labeled as unvalidated estimates or marked INSUFFICIENT EVIDENCE where no source can be cited. No measurement against a target technology library, layout, or workload has been performed for XH-1.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
|
||||||
|
The multiply-divide (MUL/DIV) unit implements the integer multiplication and division instructions defined by the M extension of the RISC-V ISA (combined with the base I extension; usually written RV64I + M, also expressed as RV64IM when the integer base is implicit). In XH-1, the MUL/DIV unit sits on the execution path of each of the 128 cores, and its latency, throughput, area, and energy directly influence per-core performance and the die-level power and thermal envelope. This document surveys existing MUL/DIV architectures (iterative, array, Radix-4/8 Booth, array-of-serial, and Sutherland/Robertson / non-restoring / SRT division), compares their trade-offs across per-core and shared organizations, and proposes a design direction suitable for a 128-core RISC-V processor where replication, area, and energy are first-class constraints.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
|
||||||
|
What MUL/DIV architecture best fits XH-1's 128-core RISC-V design, given:
|
||||||
|
|
||||||
|
- Per-core area must be small enough to replicate the unit 128 times on one die, or alternatively the unit must be shared across a small cluster of cores with acceptable latency and contention.
|
||||||
|
- The pipeline stage budget for the EX stage is finite (TBD by `pipeline.md`).
|
||||||
|
- The unit is on a non-critical path for most general-purpose code but is on the critical path for scientific, cryptographic, hash, and DSP workloads.
|
||||||
|
- IEEE 754 floating-point support is out of scope for this document; only integer MUL/DIV is considered. F/D extension implications are listed as an open question.
|
||||||
|
- Verification must scale to 128 cores; deterministic, fully combinational MUL and bounded-iteration DIV designs simplify verification.
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
### Required Instructions (RV64I + M)
|
||||||
|
|
||||||
|
The M extension defines the following on RV64:
|
||||||
|
|
||||||
|
- MUL, MULH, MULHSU, MULHU: 64×64 → 128-bit multiply, returning the lower 64 bits (MUL) or the upper 64 bits (MULH and its signed/unsigned variants).
|
||||||
|
- DIV, DIVU, REM, REMU: 64-bit signed/unsigned divide and remainder.
|
||||||
|
- MULW, DIVW, REMW: 32×32 → 32-bit signed multiply and signed divide, sign-extended to 64 bits.
|
||||||
|
- DIVUW, REMUW: 32×32 → 32-bit unsigned divide, sign-extended to 64 bits.
|
||||||
|
|
||||||
|
The header of the previous revision referred to "RV64IM (and optionally M-extension)," which is incoherent because M is part of RV64IM by definition. In this document the scope is the M extension on RV64I.
|
||||||
|
|
||||||
|
### Latency and Throughput Ranges (Background Only)
|
||||||
|
|
||||||
|
Published RISC-V cores report the following approximate ranges. These are background reference values from named open cores; they are not measurements of XH-1.
|
||||||
|
|
||||||
|
| Operation | Reported latency range (cycles) | Reported throughput | Source basis |
|
||||||
|
|------------------|---------------------------------|--------------------------|---------------------------------------|
|
||||||
|
| MUL (lower 64) | 1–5 | 1/cycle (pipelined) | Rocket, BOOM Small, Ariane (claimed) |
|
||||||
|
| MULH (upper 64) | 2–6 | 1/cycle (pipelined) | Same |
|
||||||
|
| DIV/REM (64-bit) | 8–40 | 1 per N cycles | Same |
|
||||||
|
| MULW/DIVW family | similar to 64-bit variants | similar | Same |
|
||||||
|
|
||||||
|
The exact latency in any given core depends on the EX-stage timing budget, the process corner, and the divider radix. INSUFFICIENT EVIDENCE exists in the XH-1 repository to pin XH-1 to a specific cycle count.
|
||||||
|
|
||||||
|
### Division Algorithms (Background)
|
||||||
|
|
||||||
|
- Restoring division: simple, one bit per iteration, simple control, slow.
|
||||||
|
- Non-restoring division: similar iteration count, slightly smaller area than the array form, well-suited to iterative implementation.
|
||||||
|
- SRT division (radix-2, radix-4, radix-16): quotient-digit selection allows more than one bit per iteration. Higher radix reduces iteration count at the cost of more complex selection logic and a larger redundant residual representation.
|
||||||
|
- Newton–Raphson reciprocal multiplication: pre-computes an approximation of the reciprocal via iteration, then multiplies. Low latency for many operand classes, but variable latency and convergence-dependent.
|
||||||
|
- Goldschmidt division: similar trade-off to Newton–Raphson.
|
||||||
|
|
||||||
|
### Multiplication Algorithms (Background)
|
||||||
|
|
||||||
|
- Shift-and-add multiplier: simple, slow, one bit per cycle.
|
||||||
|
- Array (Braun / Baugh–Wooley) multiplier: combinational, O(n²) partial products and full-adders, deterministic single-cycle latency, large area.
|
||||||
|
- Wallace / Dadda tree multiplier: O(n log n) partial-product reduction using 3:2 and 4:2 compressors; faster critical path, less regular layout.
|
||||||
|
- Booth-encoded multipliers (radix-4, radix-8): reduce partial-product count by roughly 2× (radix-4) or 3× (radix-8) compared to a naive array.
|
||||||
|
- Compressor-tree implementations (3:2 counters, 4:2 compressors, Ling / Han–Carlson adders) trade area for shorter critical paths.
|
||||||
|
|
||||||
|
## Existing Approaches (Per-Core, Unless Noted)
|
||||||
|
|
||||||
|
The following architectures are well-known and serve as comparison baselines. All area-multiplier numbers in this section are unvalidated estimates; no synthesis has been performed.
|
||||||
|
|
||||||
|
### A1. Iterative Shift-and-Add Multiplier + Restoring Divider
|
||||||
|
|
||||||
|
- Multiplier: 64 cycles, 1 bit/cycle.
|
||||||
|
- Divider: 64 cycles (restoring) or ~64 cycles (non-restoring).
|
||||||
|
- Area: very small. Used as the 1× baseline in the comparison table.
|
||||||
|
|
||||||
|
### A2. Booth-Radix-4 Multiplier + Iterative Non-Restoring Divider
|
||||||
|
|
||||||
|
- Multiplier: 32 radix-4 digits → 33 partial products (including the sign-extension row) reduced through a Wallace/Dadda or 4:2 tree. Two pipeline stages; 2-cycle latency, 1/cycle throughput.
|
||||||
|
- Divider: ~64-cycle non-restoring, 1/64 throughput.
|
||||||
|
- Area: estimated ~3–4× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### A3. Pipelined Radix-4/8 Booth Multiplier + Radix-4 SRT Divider
|
||||||
|
|
||||||
|
- Multiplier: 2–3 stage pipeline, 1/cycle issue, latency 2–3 cycles.
|
||||||
|
- Divider: radix-4 SRT, ~16 cycle latency. If pipelined internally, throughput approaches 1/cycle on independent operands; if not pipelined, throughput is 1/16. The SRT-4 in this class is typically not deeply pipelined in published small RV cores.
|
||||||
|
- Area: estimated ~5–7× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### A4. Fully Combinational Array Multiplier + Radix-16 SRT Divider
|
||||||
|
|
||||||
|
- Multiplier: 64×64 → 128 in a single cycle, large area.
|
||||||
|
- Divider: 4–8 cycle latency via radix-16 selection.
|
||||||
|
- Area: estimated ~10–15× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### A5. Shared / Clustered MUL/DIV Unit
|
||||||
|
|
||||||
|
- A single MUL/DIV unit serves N cores, accessed via a reservation station or blocking interface.
|
||||||
|
- Saves replicated area; increases latency under contention.
|
||||||
|
- A natural fit for 128-core designs only if inter-core divide/mul traffic is low or bursty.
|
||||||
|
|
||||||
|
This alternative was not analyzed in depth in the previous revision; it is included here as a candidate organization.
|
||||||
|
|
||||||
|
## Alternative Designs (Per-Core, Unless Noted)
|
||||||
|
|
||||||
|
### B1. 2-Stage Pipelined Radix-4 Booth Multiplier
|
||||||
|
|
||||||
|
- Radix-4 Booth encoding of two 64-bit operands produces 33 partial products (32 signed digit products plus one sign-correction row). The previous revision's "17 partial products" figure was incorrect and is corrected here.
|
||||||
|
- Partial-product reduction through a 4:2 compressor tree, followed by a final carry-propagate add split across two pipeline stages.
|
||||||
|
- Latency 2 cycles, throughput 1/cycle.
|
||||||
|
- Divider: 32-cycle radix-2 non-restoring iterative divider with explicit worst-case latency; "early-exit" optimizations (e.g., leading-zero detection on the divisor, trailing-zero detection on the dividend to shift out low bits before the loop) reduce typical-case latency but the worst case remains 32 cycles. The previous revision named the optimization without describing it; the description is provided here.
|
||||||
|
- Area: estimated ~3–5× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### B2. 3-Stage Pipelined Radix-8 Booth Multiplier
|
||||||
|
|
||||||
|
- 11 partial products for a 64-bit operand (one per 3-bit window plus sign correction).
|
||||||
|
- Latency 3 cycles, throughput 1/cycle.
|
||||||
|
- Smaller critical path than B1 at the cost of higher area and more complex Booth-3 encoding.
|
||||||
|
|
||||||
|
### B3. Iterative Multiplier with Multi-Cycle Variable Latency
|
||||||
|
|
||||||
|
- A single 64×64 multiplier reused across MUL/MULH/MULHSU/MULHU by selecting output bits.
|
||||||
|
- Latency 4–5 cycles, 1/cycle throughput.
|
||||||
|
- Smaller area than B1; longer issue-to-use distance complicates scheduling.
|
||||||
|
|
||||||
|
### B4. Pipelined Radix-4 SRT Divider
|
||||||
|
|
||||||
|
- 16-cycle latency when issued back-to-back with different operands. The previous revision was internally inconsistent (text said fully pipelined, table said 1/16 throughput). For this document, B4 is treated as a 16-cycle-latency, 1/16-throughput non-pipelined unit; a fully pipelined SRT-4 with 1/cycle throughput would add ~3–4× the divider area and is a separate design point (B4-pipe).
|
||||||
|
- Larger area than B1's divider, more verification complexity (quotient-digit selection must be proven correct for all residuals).
|
||||||
|
|
||||||
|
### B5. Newton–Raphson Divider with Hardware Reciprocal Iteration
|
||||||
|
|
||||||
|
- ~4–6 cycles for 32-bit reciprocal convergence on typical operands, then one multiply.
|
||||||
|
- Lowest divide latency for many operand classes, but variable and dependent on operand class (convergence count is worst-case bounded but typical-case data-dependent).
|
||||||
|
- Complex verification: requires convergence proof or guarded iteration count plus a fallback path.
|
||||||
|
|
||||||
|
### B6. Two's-Complement Array Divider (Combinational)
|
||||||
|
|
||||||
|
- Single-cycle 64-bit divide.
|
||||||
|
- Estimated very large area (order-of-magnitude larger than a 64-bit combinational multiplier, but INSUFFICIENT EVIDENCE for an exact ratio). The previous revision's "~50×" figure was an unsupported round number; the document does not retain it.
|
||||||
|
- Likely unacceptable for 128-core replication.
|
||||||
|
|
||||||
|
### B7. Cluster-Shared MUL/DIV Unit (8 cores per unit, 16 units total)
|
||||||
|
|
||||||
|
- A single radix-4 Booth + radix-4 SRT divider (B1 + B4) shared by 8 cores via a small reservation station.
|
||||||
|
- 16 instances on the die instead of 128.
|
||||||
|
- Per-cluster area budget can absorb a faster divider (e.g., B4-pipe or B5) than per-core replication allows.
|
||||||
|
- Latency and contention penalty under simultaneous divide requests.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
The following table lists estimated per-unit characteristics. All numbers are unvalidated estimates pending synthesis. Relative area is normalized to A1 (1×); the actual ratios depend on the technology library, target frequency, and choice of final adder.
|
||||||
|
|
||||||
|
| Design | MUL Latency | MUL Throughput | DIV Latency | DIV Throughput | Relative Area (per unit) | Verification Complexity |
|
||||||
|
|-------------------------------------------------|-------------|----------------|-------------|----------------|--------------------------|-------------------------|
|
||||||
|
| A1. Shift-add + restoring | 64 | 1/64 | 64 | 1/64 | 1× | Low |
|
||||||
|
| A2. Booth-r4 + NR div | 2 | 1/1 | ~64 | 1/64 | ~3–4× (unvalidated) | Low–Medium |
|
||||||
|
| A3. Pipelined r4/r8 + SRT-4 (non-pipe) | 2–3 | 1/1 | 16 | 1/16 | ~5–7× (unvalidated) | Medium |
|
||||||
|
| A4. Combinational + SRT-16 | 1 | 1/1 | 4–8 | 1/4–1/8 | ~10–15× (unvalidated) | High |
|
||||||
|
| B1. 2-stage r4 + NR-32 (with early-exit) | 2 | 1/1 | 32 worst | 1/32 | ~3–5× (unvalidated) | Low–Medium |
|
||||||
|
| B2. 3-stage r8 + NR-32 | 3 | 1/1 | 32 worst | 1/32 | ~4–6× (unvalidated) | Medium |
|
||||||
|
| B4. SRT-4 (non-pipelined) | n/a | n/a | 16 | 1/16 | adds ~3–4× to A2 (unvalidated) | Medium |
|
||||||
|
| B4-pipe. SRT-4 (pipelined, 1/cycle) | n/a | n/a | 16 | 1/1 | adds ~6–8× to A2 (unvalidated) | Medium–High |
|
||||||
|
| B5. Newton–Raphson | 1 | 1/1 | 4–6 | 1/4–1/6 | ~6–8× (unvalidated) | High |
|
||||||
|
| B6. Combinational array divider | 1 | 1/1 | 1 | 1/1 | very large (unvalidated) | Medium |
|
||||||
|
| B7. Cluster-shared (per 8 cores) B1 + B4-pipe | 2 | 1/1 | 16 | 1/1 (when free) | per-cluster ~8–12× A1; ×16 instances (unvalidated) | Medium–High |
|
||||||
|
|
||||||
|
The previous revision compared only per-core designs. The revised comparison adds B7 and notes the cluster organization explicitly, addressing the prior omission.
|
||||||
|
|
||||||
|
## Proposal (Resolved)
|
||||||
|
|
||||||
|
PROPOSAL: Adopt B1 (2-stage pipelined Radix-4 Booth multiplier, 33 partial products, 4:2 compressor tree, 2-cycle latency, 1/cycle throughput) paired with a 32-cycle radix-2 non-restoring iterative divider with explicit worst-case latency and an early-exit optimization for typical operands.
|
||||||
|
|
||||||
|
This resolves the prior contradiction between the abstract PROPOSAL (which named a pipelined SRT-4) and the final Recommendation (which named the 32-cycle non-restoring divider). The remaining sections are aligned to B1 + non-restoring 32-cycle divider.
|
||||||
|
|
||||||
|
Conditional variant: If early workload analysis (TBD) shows that 32-cycle divide latency is a bottleneck, upgrade the divider to B4-pipe (pipelined radix-4 SRT, 16-cycle latency, 1/cycle throughput) at an estimated additional ~3–4× divider area (unvalidated). This is a design knob, not a parallel recommendation.
|
||||||
|
|
||||||
|
Open variant: B7 (cluster-shared organization) is retained as a fallback if per-core area constraints prove tighter than estimated. The decision requires quantitative synthesis data that is not yet available.
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
|
||||||
|
- 2-cycle pipelined Radix-4 Booth multiplier offers 1/cycle throughput at modest per-core area, suitable for replicated 128-core operation.
|
||||||
|
- 32-cycle non-restoring iterative divider (with early-exit for typical operands) meets the latency needs of general-purpose code without dominating area.
|
||||||
|
- Deterministic latencies simplify pipeline scheduling, forwarding, and verification compared to Newton–Raphson.
|
||||||
|
- Radix-4 Booth and radix-2 non-restoring division are well-understood and have reference implementations in Rocket, BOOM, Ariane, and other open-source RISC-V cores (specific measurements: INSUFFICIENT EVIDENCE in the XH-1 repository).
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
|
||||||
|
- 32-cycle divide is slow for workloads dominated by large-integer arithmetic (RSA, big-integer math, certain cryptographic primitives). The conditional B4-pipe upgrade addresses this at additional area cost.
|
||||||
|
- Newton–Raphson dividers reduce typical latency for many operand classes but at unacceptable per-core verification cost for 128-core replication in the current design envelope.
|
||||||
|
- A non-restoring iterative divider has the lowest per-core area but penalizes every divide by up to 32 cycles, which can be felt in hash-table probing, parser/lexer dispatch, and some interpreter dispatch loops.
|
||||||
|
- Booth encoding complicates verification of signed/unsigned correctness for MULH / MULHSU / MULHU; the previous revision flagged this; it is reaffirmed here.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
|
||||||
|
- ASSUMPTION: XH-1 is a 128-core design with a short pipeline (TBD by `pipeline.md`). The MUL/DIV unit must fit in a small per-core area budget and the EX-stage latency budget.
|
||||||
|
- ASSUMPTION: The design targets a balance of general-purpose and HPC-adjacent workloads; therefore a moderately fast (but not the fastest) divider is acceptable as the default, with an explicit upgrade path.
|
||||||
|
- The multiplier is sized to produce the full 128-bit product so that MULH/MULHU selection requires only output muxing, not a separate datapath.
|
||||||
|
- The W-variants (MULW, DIVW, DIVUW, REMW, REMUW) reuse the lower 32 bits of the 64-bit datapath with sign-extension at the output, not a separate 32-bit datapath.
|
||||||
|
|
||||||
|
### Instruction Coverage
|
||||||
|
|
||||||
|
- MUL/MULH/MULHSU/MULHU: supported.
|
||||||
|
- DIV/DIVU/REM/REMU: supported, with RISC-V-spec corner cases handled explicitly.
|
||||||
|
- MULW/DIVW/DIVUW/REMW/REMUW: supported via shared datapath.
|
||||||
|
|
||||||
|
### Edge Cases (RISC-V Spec)
|
||||||
|
|
||||||
|
The following are the architecturally specified results. The previous revision contained an error in the DIVU-by-zero description; the corrected behavior is:
|
||||||
|
|
||||||
|
- DIV by zero: quotient = −1 (all bits set in the lower XLEN), remainder = dividend (x).
|
||||||
|
- DIVU by zero: quotient = 2^XLEN − 1 (all ones), remainder = dividend (x).
|
||||||
|
- REM by zero: remainder = dividend (x); quotient = −1.
|
||||||
|
- REMU by zero: remainder = dividend (x); quotient = 2^XLEN − 1.
|
||||||
|
- Signed overflow (DIV of INT64_MIN by −1): quotient = INT64_MIN, remainder = 0.
|
||||||
|
- REM sign rule: the sign of the remainder follows the sign of the dividend. This is a frequent bug source and must be covered explicitly in verification; the previous revision did not call it out.
|
||||||
|
|
||||||
|
The unit must produce these results without raising an exception.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
|
||||||
|
- Per-core area: a small 64×64 → 128-bit Radix-4 Booth multiplier with a 4:2 compressor tree, a 2-stage pipelined final adder, and a 32-cycle non-restoring divider is expected to be a small fraction of a typical RV core area, but the exact fraction is INSUFFICIENT EVIDENCE because no baseline core area has been established for XH-1. The previous revision's "<1–2% of a typical small RV core area" is removed because it lacked a baseline.
|
||||||
|
- Floorplanning: a regular MUL/DIV layout that mirrors across all 128 cores is preferred to avoid routing asymmetry that would break clock distribution and thermal symmetry.
|
||||||
|
- Voltage / frequency: a moderately pipelined MUL/DIV is more resilient to voltage droop; this is an advantage for 128-core operation.
|
||||||
|
- Contention: per-core replication means there is no inter-core contention for the MUL/DIV unit. The only contention is intra-core (e.g., two dependent divides in flight in an out-of-order pipeline). The cluster-shared B7 organization reintroduces inter-core contention; this is the central trade-off.
|
||||||
|
- Test / DFT: 128 instances of the MUL/DIV unit (or 16 instances in B7) must be tested. A scan-friendly, fully synchronous design with no asynchronous reset paths inside the iterative divider is preferable. DFT strategy is addressed in a dedicated section below.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
|
||||||
|
- For general-purpose code, MUL/DIV is rarely the bottleneck; a 2-cycle MUL latency matches typical issue-to-use distances.
|
||||||
|
- For cryptography, the relevant metric is MUL (lower 64) throughput, since Karatsuba and Montgomery multiplication depend on lower-half multiplies. The previous revision's claim that MULH throughput is the key for Karatsuba / Montgomery is not generally correct and is corrected here. MULH is the key metric for software-emulated 128-bit integer multiplication (`__int128`), not for Karatsuba / Montgomery in the typical formulation.
|
||||||
|
- For hash tables and interpreters, DIV latency matters more than throughput; a 32-cycle divider is acceptable but not ideal.
|
||||||
|
- For 32-bit integer code, the W-variants are the hot path; reusing the 64-bit datapath with muxed operands is acceptable.
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
|
||||||
|
- ASSUMPTION (unvalidated): A 64×64 → 128-bit Radix-4 Booth multiplier with a 4:2 compressor tree and 2-stage pipelined final adder occupies roughly 0.02–0.10 mm² in a typical 7nm process, depending on target frequency, Vdd, and FF corner. No source is cited; INSUFFICIENT EVIDENCE to refine this without a technology file.
|
||||||
|
- ASSUMPTION (unvalidated): A 32-cycle non-restoring iterative divider occupies roughly 0.01–0.05 mm² in the same envelope.
|
||||||
|
- PROPOSAL (unvalidated): Total MUL/DIV area target: <0.15 mm² per core. 128× replication: ~20 mm². These numbers are order-of-magnitude estimates and must be replaced with synthesis data before being used for die budgeting.
|
||||||
|
- The 4:2 compressor tree is a common implementation choice for radix-4 reduction. The previous revision's claim that it is "more area-efficient than a Wallace tree" is removed; Wallace and Dadda trees implemented with 4:2 compressors are not generally distinguishable in area for radix-4 reduction, and the choice is largely a layout / regularity preference. INSUFFICIENT EVIDENCE to prefer one over the other without synthesis.
|
||||||
|
- Sign-extension and zero-extension muxes for MULW are negligible area.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
|
||||||
|
- A 64×64 multiplier tree has high switching activity; clock gating when the unit is idle is essential.
|
||||||
|
- The iterative divider has lower average switching power than a fully combinational divider, but its long residency increases leakage energy per operation.
|
||||||
|
- Power gating: at 128 cores, a per-core power-gate for the MUL/DIV unit is worth considering if idle periods dominate. The wake-up latency and IR-drop impact on the power grid are not yet analyzed; INSUFFICIENT EVIDENCE without a full-die power analysis.
|
||||||
|
- Energy per multiply is dominated by the compressor tree. The relationship "energy scales roughly with the number of partial products" is a rule of thumb; INSUFFICIENT EVIDENCE for a specific quantitative claim.
|
||||||
|
- Energy per divide is dominated by the 32-cycle residency; early-exit on the dividend (trailing-zero detection) reduces energy for typical operands but the worst-case energy remains.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
|
||||||
|
- Synchronous, single-clock-domain design inside the unit.
|
||||||
|
- No asynchronous resets inside the iterative divider; synchronous reset only at the start of an operation.
|
||||||
|
- Final carry-propagate adder: Kogge–Stone, Han–Carlson, and Brent–Kung are all viable. The choice depends on the EX-stage timing budget and the area target; Kogge–Stone / Han–Carlson are faster but larger, Brent–Kung is smaller but slower. The previous revision named a default without justification; this document records the choice as a trade-off driven by the EX-stage timing budget, to be decided after synthesis.
|
||||||
|
- Divider quotient and remainder registers are 64 bits each, with an extra bit for the iterative sign.
|
||||||
|
- The microarchitectural state machine is small and well-suited to a one-hot or binary-encoded FSM.
|
||||||
|
- Output muxing for MUL / MULH / MULHSU / MULHU / MULW is a small mux tree, not a separate datapath.
|
||||||
|
- Booth encoding produces 33 partial products for radix-4 of a 64-bit operand (32 digit products plus one sign-correction row). The "17 partial products" figure from the previous revision is incorrect and is corrected here.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
|
||||||
|
- Formal verification of the multiplier compressor tree and final adder is feasible with bounded model checkers and is recommended.
|
||||||
|
- Directed tests for division edge cases: ÷0 (signed and unsigned, both quotient and remainder), INT64_MIN / −1, dividend = divisor, dividend = 0, divisor = 1, all-ones, alternating bits, dividend = −1, divisor = 2.
|
||||||
|
- Explicit coverage of the REM sign-of-dividend rule.
|
||||||
|
- Coverage of all 4 MUL variants and the W-variants.
|
||||||
|
- Randomized differential testing against a software reference (a GCC-compiled test harness running on Spike or QEMU, or a Python / C++ golden model) is recommended.
|
||||||
|
- 128-core DFT: see the dedicated section below.
|
||||||
|
|
||||||
|
### DFT Strategy (128 Replicated Units)
|
||||||
|
|
||||||
|
- Single-clock-domain, synchronous-reset-only design is required for scan insertion.
|
||||||
|
- Each MUL/DIV instance is scan-stitched independently; long scan chains between the iterative divider and surrounding logic are avoided to prevent hold-time issues.
|
||||||
|
- Scan compression: per-core compression reduces the number of top-level scan pins; the compression architecture is TBD by the DFT plan.
|
||||||
|
- BIST: optional per-core BIST for the MUL/DIV unit is feasible given its small size and regular structure; this would reduce ATPG complexity at the cost of additional area for the BIST controller.
|
||||||
|
- ATPG implications: the iterative divider is the only sequential element of consequence in the MUL/DIV block; full-scan coverage is straightforward if no asynchronous paths are introduced.
|
||||||
|
|
||||||
|
## Software Implications
|
||||||
|
|
||||||
|
- Compilers emit MUL freely; no software changes are required.
|
||||||
|
- Division by a constant is often transformed by the compiler into a magic-number multiply; the choice of MUL/DIV design therefore disproportionately affects runtime divide performance for code with frequent constant divides.
|
||||||
|
- For languages with software-emulated 128-bit integers (`__int128` in C/C++), the compiler emits MULH / MULHU sequences; MULH latency and throughput are the relevant metrics here. This is the case where MULH is the key metric, not the crypto case.
|
||||||
|
- Crypto libraries (libsodium, OpenSSL, mbedTLS) typically use Karatsuba or Montgomery multiplication, which depend on lower-half MUL throughput, not MULH throughput. The previous revision's claim that MULH is the key for Karatsuba / Montgomery is corrected here.
|
||||||
|
- JavaScript engines and language runtimes may issue frequent DIVs for tagged-value unpacking; a 32-cycle divider increases interpreter dispatch latency for that pattern.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
|
||||||
|
RECOMMENDATION: Adopt B1 (2-stage pipelined Radix-4 Booth multiplier, 33 partial products, 4:2 compressor tree, 2-cycle latency, 1/cycle throughput) paired with a 32-cycle radix-2 non-restoring iterative divider with explicit worst-case latency and an early-exit optimization for typical operands.
|
||||||
|
|
||||||
|
Rationale:
|
||||||
|
|
||||||
|
- Best balance of area, energy, and performance for 128-core replication under current unvalidated estimates.
|
||||||
|
- Deterministic, fully synchronous design simplifies verification and DFT.
|
||||||
|
- 1/cycle MUL throughput meets general-purpose and crypto lower-half-multiply needs.
|
||||||
|
- 32-cycle worst-case DIV latency is acceptable for general-purpose workloads; an upgrade path to B4-pipe (16-cycle pipelined SRT-4) is reserved for the case where profiling evidence supports it.
|
||||||
|
|
||||||
|
RECOMMENDATION (conditional): If early workload analysis (TBD) shows that divide latency is a bottleneck, upgrade the divider to B4-pipe (pipelined Radix-4 SRT, 16-cycle latency, 1/cycle throughput) at an estimated additional ~3–4× divider area (unvalidated). Do not adopt a Newton–Raphson divider unless profiling evidence strongly supports it, due to verification complexity and variable latency.
|
||||||
|
|
||||||
|
RECOMMENDATION: Do not adopt a fully combinational array divider or fully combinational multiplier for the replicated 128-core design. The area cost is not justified by the latency benefit at the per-core replication factor.
|
||||||
|
|
||||||
|
RECOMMENDATION (fallback): If per-core area constraints prove tighter than estimated, evaluate B7 (cluster-shared organization, 16 instances serving 8 cores each) with a faster per-cluster divider (B4-pipe or B5) before reducing the per-core divider latency further. This trades inter-core contention for per-core area and is the right knob to pull when the per-core area budget is the binding constraint.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
|
||||||
|
- Direction of recommendation: MEDIUM. The general design class is well-established; the specific B1 + non-restoring-32-cycle choice depends on a per-core area budget that has not yet been validated.
|
||||||
|
- Quantitative area, latency, power, and energy numbers: LOW. No measurements exist in the XH-1 repository; all such numbers in this document are explicitly labeled as unvalidated estimates or INSUFFICIENT EVIDENCE.
|
||||||
|
- Verification strategy: HIGH. The recommended approach (formal on the multiplier, directed + randomized for the divider, explicit REM sign rule) is standard practice.
|
||||||
|
- Edge-case correctness: HIGH. The RISC-V spec behavior is explicit; the previous revision contained an error in the DIVU-by-zero description, which is corrected here.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- What is the EX-stage latency budget? (Depends on `pipeline.md`.)
|
||||||
|
- What is the target frequency and process node? Determines whether 2-cycle MUL is feasible, and which final-adder architecture is appropriate.
|
||||||
|
- Is the design in-order or out-of-order? Out-of-order execution can hide divide latency; in-order cannot.
|
||||||
|
- Will the MUL/DIV unit share an issue port with the ALU, or have a dedicated issue port? (Likely shared, but TBD.)
|
||||||
|
- Is there a future F / D extension? If so, the integer MUL/DIV unit may also need to feed FP-to-int conversions or FP reciprocal iterations; this is not yet analyzed.
|
||||||
|
- What is the expected workload mix? General-purpose vs. HPC vs. embedded vs. server?
|
||||||
|
- Should the MUL/DIV unit be power-gated when idle? At 128 cores, idle probability may be high; wake-up latency and IR-drop impact are TBD.
|
||||||
|
- Should the divider support a "fast-path" for division by a small constant (e.g., a compiler-inserted reciprocal-multiply hint) as a microarchitectural feature?
|
||||||
|
- Is the B7 cluster-shared organization a realistic fallback, or is per-core replication a hard requirement?
|
||||||
|
- What scan-compression architecture will be used for 128 replicated units?
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
- The RISC-V Instruction Set Manual, Volume I: Unprivileged ISA, M-extension chapter. Defines MUL, MULH, MULHSU, MULHU, DIV, DIVU, REM, REMU, and the W-variants, including the corner-case behavior for division by zero and signed overflow. Specific URL and section: INSUFFICIENT EVIDENCE in the XH-1 repository to cite a specific revision; the manual is the canonical source.
|
||||||
|
- Hennessy and Patterson, "Computer Architecture: A Quantitative Approach." General MUL/DIV trade-off discussion.
|
||||||
|
- Parhami, "Computer Arithmetic: Algorithms and Hardware Designs." Standard reference for multiplier and divider algorithms including Booth, Wallace / Dadda, SRT, and Newton–Raphson.
|
||||||
|
- Open-source RISC-V cores: Rocket Chip, BOOM (SmallBoomConfig and MediumBoomConfig), Ariane. Used as design-class references for typical MUL/DIV organizations and reported latency ranges. Specific commits, measured numbers, and PPA data: INSUFFICIENT EVIDENCE in the XH-1 repository to cite.
|
||||||
|
- No quantitative claim in this document is derived from a measurement of XH-1 silicon, layout, or synthesis. All such numbers are explicitly labeled as unvalidated estimates or INSUFFICIENT EVIDENCE.
|
||||||
+301
File diff suppressed because one or more lines are too long
+339
@@ -0,0 +1,339 @@
|
|||||||
|
# Multiply-Divide Unit
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Engineering research document, pre-synthesis. All quantitative values below are explicitly labeled as unvalidated estimates, heuristics, or INSUFFICIENT EVIDENCE where no source can be cited. No measurement against a target technology library, layout, or workload has been performed for XH-1. The primary recommendation is conditional on synthesis data that does not yet exist in the XH-1 repository.
|
||||||
|
|
||||||
|
## Abstract
|
||||||
|
|
||||||
|
The multiply-divide (MUL/DIV) unit implements the integer multiplication and division instructions defined by the M extension of the RISC-V ISA on the RV64I base (commonly written RV64IM, since the M extension is implied when one says RV64I + M). In XH-1, the MUL/DIV unit sits on the execution path of each of the 128 cores, and its latency, throughput, area, and energy directly influence per-core performance and the die-level power and thermal envelope. This document surveys existing MUL/DIV architectures (iterative, array, Radix-4/8 Booth, array-of-serial, non-restoring, and SRT division), compares their trade-offs across per-core and cluster-shared organizations, and proposes a design direction suitable for a 128-core RISC-V processor where replication, area, and energy are first-class constraints. No design point in this document has been validated against a target process, frequency, or workload.
|
||||||
|
|
||||||
|
## Research Question
|
||||||
|
|
||||||
|
What MUL/DIV architecture best fits XH-1's 128-core RISC-V design, given:
|
||||||
|
|
||||||
|
- Per-core area must be small enough to replicate the unit 128 times on one die, or alternatively the unit must be shared across a small cluster of cores with acceptable latency and contention. The per-core area budget is currently UNKNOWN; the EX-stage latency budget is TBD by `pipeline.md`.
|
||||||
|
- The unit is on a non-critical path for most general-purpose code but is on the critical path for scientific, cryptographic, hash, and DSP workloads.
|
||||||
|
- IEEE 754 floating-point support is out of scope for this document; only integer MUL/DIV is considered. F/D extension implications are listed as an open question.
|
||||||
|
- Verification must scale to 128 cores; deterministic, fully combinational MUL and bounded-iteration DIV designs simplify verification. A divider with data-dependent early-exit is a verification complication that is noted but not resolved here.
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
### Required Instructions (RV64I + M)
|
||||||
|
|
||||||
|
The M extension defines the following on RV64 (per the RISC-V Unprivileged ISA, Document Version 20191213, Chapter 7, "M Extension"):
|
||||||
|
|
||||||
|
- MUL: 64×64 → lower 64 bits of the product.
|
||||||
|
- MULH: 64×64 signed×signed → upper 64 bits of the product.
|
||||||
|
- MULHSU: 64×64 signed (rs1) × unsigned (rs2) → upper 64 bits.
|
||||||
|
- MULHU: 64×64 unsigned×unsigned → upper 64 bits.
|
||||||
|
- DIV, DIVU, REM, REMU: 64-bit signed/unsigned divide and remainder.
|
||||||
|
- MULW: 32×32 → lower 32 bits, sign-extended to 64.
|
||||||
|
- DIVW, REMW: 32×32 signed divide and remainder, sign-extended to 64.
|
||||||
|
- DIVUW, REMUW: 32×32 unsigned divide and remainder, sign-extended to 64.
|
||||||
|
|
||||||
|
The header of the previous revision referred to "RV64IM (and optionally M-extension)," which is incoherent because M is part of RV64IM by definition. The scope of this document is the M extension on RV64I.
|
||||||
|
|
||||||
|
### Latency and Throughput Ranges (Background Only)
|
||||||
|
|
||||||
|
Published open-source RISC-V cores report approximate latency ranges. These are background reference values from named cores; they are not measurements of XH-1. The values attributed to specific cores below are taken from source-file inspection of public repositories, not from synthesis or PPA reports.
|
||||||
|
|
||||||
|
| Operation | Reported latency range (cycles) | Reported throughput | Source basis |
|
||||||
|
|------------------|---------------------------------|--------------------------|-----------------------------------------------------------------------|
|
||||||
|
| MUL (lower 64) | 1–3 | 1/cycle (pipelined) | Rocket Chip `MulDiv.scala`; Ariane `mult.sv` |
|
||||||
|
| MULH (upper 64) | 3–5 | 1/cycle (pipelined) | Rocket Chip `MulDiv.scala` |
|
||||||
|
| DIV/REM (64-bit) | 8–64 | 1 per N cycles | Rocket: 35–39 cycle radix-4 iterative; Ariane: 33–35 cycle NR-ish; BOOM: configurable |
|
||||||
|
| MULW/DIVW family | similar to 64-bit variants | similar | Same |
|
||||||
|
|
||||||
|
The exact latency in any given core depends on the EX-stage timing budget, the process corner, and the divider radix. INSUFFICIENT EVIDENCE exists in the XH-1 repository to pin XH-1 to a specific cycle count.
|
||||||
|
|
||||||
|
### Division Algorithms (Background)
|
||||||
|
|
||||||
|
- Restoring division: simple, one bit per iteration, simple control, slow.
|
||||||
|
- Non-restoring division: a class of bit-serial dividers that avoid the explicit restore step. Includes the iterative subtract-and-shift form (one bit/cycle) and array (combinational) forms. SRT is a redundant-digit extension of the non-restoring family. The term "non-restoring" in this document refers to the iterative subtract-and-shift form unless otherwise noted.
|
||||||
|
- SRT division (radix-2, radix-4, radix-16): a redundant-digit non-restoring division; quotient-digit selection allows more than one bit per iteration. Higher radix reduces iteration count at the cost of more complex selection logic and a larger redundant residual representation.
|
||||||
|
- Newton–Raphson reciprocal multiplication: pre-computes an approximation of the reciprocal via iteration, then multiplies. Variable latency, convergence-dependent.
|
||||||
|
- Goldschmidt division: similar trade-off to Newton–Raphson.
|
||||||
|
|
||||||
|
### Multiplication Algorithms (Background)
|
||||||
|
|
||||||
|
- Shift-and-add multiplier: simple, slow, one bit per cycle.
|
||||||
|
- Array (Braun / Baugh–Wooley) multiplier: combinational, O(n²) partial products and full-adders, deterministic single-cycle latency, large area.
|
||||||
|
- Wallace / Dadda tree multiplier: O(n log n) partial-product reduction using 3:2 and 4:2 compressors; faster critical path, less regular layout. The choice between Wallace and Dadda is largely a layout / regularity preference; INSUFFICIENT EVIDENCE to prefer one over the other without synthesis.
|
||||||
|
- Booth-encoded multipliers (radix-4, radix-8): reduce partial-product count compared to a naive array by recoding one operand as overlapping signed digits.
|
||||||
|
- Compressor-tree implementations (3:2 counters, 4:2 compressors, Ling / Han–Carlson adders) trade area for shorter critical paths.
|
||||||
|
|
||||||
|
### Partial-Product Counts for Radix-4 and Radix-8 Booth (Worked)
|
||||||
|
|
||||||
|
For a 64-bit signed operand encoded in radix-4 (overlapping 2-bit windows), the number of signed-digit rows is ⌈64/2⌉ = 32, plus one sign-correction row for negative-operand handling, for a total of 33 partial-product rows. The 33rd row is not a free addend; it exists to correct the sign-extension terms produced when the Booth recoding expands a negative operand. Verification authors should treat this row as a separate partial product with its own correctness argument.
|
||||||
|
|
||||||
|
For a 64-bit signed operand encoded in radix-8 (overlapping 3-bit windows), the number of signed-digit rows is ⌈64/3⌉ = 22, plus one sign-correction row, for a total of 23 partial-product rows. The previous revision's "11 partial products" figure for a 64-bit radix-8 encoder is incorrect and is corrected here; 11 is approximately correct for a 32-bit radix-8 encoding (⌈32/3⌉ = 11 digits, no separate sign row in some formulations) but not for 64-bit.
|
||||||
|
|
||||||
|
## Existing Approaches (Per-Core, Unless Noted)
|
||||||
|
|
||||||
|
The following architectures are well-known and serve as comparison baselines. All area-multiplier numbers in this section are unvalidated estimates; no synthesis has been performed.
|
||||||
|
|
||||||
|
### A1. Iterative Shift-and-Add Multiplier + Restoring Divider
|
||||||
|
|
||||||
|
- Multiplier: 64 cycles, 1 bit/cycle.
|
||||||
|
- Divider: 64 cycles (restoring) or ~64 cycles (non-restoring). A 64-bit radix-2 non-restoring divider takes 64 iterations on a 64-bit operand; the 32-cycle figure in the previous revision is unjustified for a 64-bit radix-2 design and is corrected to 64 here.
|
||||||
|
- Area: very small. Used as the 1× baseline in the comparison table.
|
||||||
|
|
||||||
|
### A2. Booth-Radix-4 Multiplier + Iterative Non-Restoring Divider
|
||||||
|
|
||||||
|
- Multiplier: 33 partial products reduced through a 4:2 tree. Two pipeline stages; 2-cycle latency, 1/cycle throughput.
|
||||||
|
- Divider: 64-cycle non-restoring, 1/64 throughput.
|
||||||
|
- Area: estimated ~3–4× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### A3. Pipelined Radix-4/8 Booth Multiplier + Radix-4 SRT Divider (Non-Pipelined SRT)
|
||||||
|
|
||||||
|
- Multiplier: 2–3 stage pipeline, 1/cycle issue, latency 2–3 cycles.
|
||||||
|
- Divider: radix-4 SRT, ~16 cycle latency when not pipelined, 1/16 throughput. The previous revision oscillated between "pipelined internally" and "typically not deeply pipelined"; this entry adopts the non-pipelined SRT-4 view for direct comparison with B4. A pipelined SRT-4 with 1/cycle throughput is a separate design point (B4-pipe).
|
||||||
|
- Area: estimated ~5–7× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### A4. Fully Combinational Array Multiplier + Radix-16 SRT Divider
|
||||||
|
|
||||||
|
- Multiplier: 64×64 → 128 in a single cycle, large area.
|
||||||
|
- Divider: 4–8 cycle latency via radix-16 selection.
|
||||||
|
- Area: estimated ~10–15× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### A5. Shared / Clustered MUL/DIV Unit
|
||||||
|
|
||||||
|
- A single MUL/DIV unit serves N cores, accessed via a small FIFO or blocking interface.
|
||||||
|
- Saves replicated area; increases latency under contention.
|
||||||
|
- A natural fit for 128-core designs only if inter-core divide/mul traffic is low or bursty.
|
||||||
|
|
||||||
|
This alternative was not analyzed in depth in the previous revision; it is included here as a candidate organization (see B7).
|
||||||
|
|
||||||
|
## Alternative Designs (Per-Core, Unless Noted)
|
||||||
|
|
||||||
|
### B1. 2-Stage Pipelined Radix-4 Booth Multiplier + 64-Cycle Non-Restoring Divider
|
||||||
|
|
||||||
|
- Radix-4 Booth encoding of two 64-bit operands produces 33 partial products (32 signed-digit rows plus one sign-correction row). Partial-product reduction through a 4:2 compressor tree, followed by a final carry-propagate add split across two pipeline stages.
|
||||||
|
- Latency 2 cycles, throughput 1/cycle.
|
||||||
|
- Divider: 64-cycle radix-2 non-restoring iterative divider with explicit worst-case latency. The previous revision named 32 cycles; the corrected value is 64 iterations for a 64-bit operand, with the option of a 1-cycle "skip when both operands are zero" guard but no other early-exit that would change the worst case. Any data-dependent early-exit (e.g., trailing-zero detection on the dividend) is recorded as a TBD microarchitectural feature and not assumed in the worst-case latency budget.
|
||||||
|
- Area: estimated ~3–5× of A1; INSUFFICIENT EVIDENCE without a technology file.
|
||||||
|
|
||||||
|
### B2. 3-Stage Pipelined Radix-8 Booth Multiplier + 64-Cycle Non-Restoring Divider
|
||||||
|
|
||||||
|
- 23 partial products for a 64-bit operand (22 signed-digit rows plus one sign-correction row). The previous revision's "11 partial products" figure is corrected here.
|
||||||
|
- Latency 3 cycles, throughput 1/cycle.
|
||||||
|
- Smaller critical path than B1 at the cost of higher area and more complex Booth-3 encoding.
|
||||||
|
|
||||||
|
### B3. Iterative Multiplier with Multi-Cycle Variable Latency
|
||||||
|
|
||||||
|
- A single 64×64 multiplier reused across MUL/MULH/MULHSU/MULHU by selecting output bits.
|
||||||
|
- Latency 4–5 cycles, 1/cycle throughput.
|
||||||
|
- Smaller area than B1; longer issue-to-use distance complicates scheduling.
|
||||||
|
|
||||||
|
### B4. Pipelined Radix-4 SRT Divider (Standalone, Two Variants)
|
||||||
|
|
||||||
|
- B4 (non-pipelined): 16-cycle latency, 1/16 throughput, non-pipelined.
|
||||||
|
- B4-pipe (pipelined): 16-cycle latency, 1/cycle throughput, deeply pipelined.
|
||||||
|
- The previous revision was internally inconsistent between text and table; the two variants are recorded here as separate design points.
|
||||||
|
- Divider-only area: estimated ~3–4× the B1 non-restoring divider. When added to a B1 multiplier, total unit area is estimated ~6–8× of A1 for B4-pipe. The "3–4× divider" and "6–8× total to A2" numbers in the previous revision refer to different baselines and are reconciled in the Comparison table.
|
||||||
|
- Larger area than B1's divider; more verification complexity (quotient-digit selection must be proven correct for all residuals).
|
||||||
|
|
||||||
|
### B5. Newton–Raphson Divider with Hardware Reciprocal Iteration (64-bit)
|
||||||
|
|
||||||
|
- For 64-bit dividend/divisor: a 32-bit reciprocal approximation is refined via Newton–Raphson iteration (typically 2 iterations to reach 64-bit accuracy), then multiplied by the dividend, with a correction step. Total latency is approximately 6–10 cycles for 64-bit operands, of which the multiplies are 1 cycle each (assuming a B1-class multiplier is available) and the reciprocal iterations are 1–2 cycles each. The 4–6 cycle figure in the previous revision applies to 32-bit operands and is corrected here for the 64-bit case.
|
||||||
|
- Lowest divide latency for many operand classes, but variable and dependent on operand class (convergence count is worst-case bounded but typical-case data-dependent).
|
||||||
|
- Complex verification: requires convergence proof or guarded iteration count plus a fallback path.
|
||||||
|
|
||||||
|
### B6. Combinational Array Divider (Non-Restoring 2D Cell Array)
|
||||||
|
|
||||||
|
- Single-cycle 64-bit divide via a 2D array of controlled add/subtract cells, sometimes called a "combinational non-restoring divider" or just "array divider." The term is not standardized; this document uses "combinational non-restoring array divider" to disambiguate.
|
||||||
|
- Estimated very large area (order-of-magnitude larger than a 64-bit combinational multiplier, but INSUFFICIENT EVIDENCE for an exact ratio). The previous revision's "~50×" figure was an unsupported round number and is not retained.
|
||||||
|
- Likely unacceptable for 128-core replication.
|
||||||
|
|
||||||
|
### B7. Cluster-Shared MUL/DIV Unit (8 Cores per Unit, 16 Units Total)
|
||||||
|
|
||||||
|
- A single radix-4 Booth multiplier + radix-4 SRT divider (B1 + B4-pipe) shared by 8 cores via a small FIFO.
|
||||||
|
- 16 instances on the die instead of 128.
|
||||||
|
- Per-cluster area budget can absorb a faster divider than per-core replication allows.
|
||||||
|
- Latency and contention penalty under simultaneous divide requests. Contention behavior depends on the issue model (blocking, non-blocking with FIFO, full reservation station); the choice is TBD.
|
||||||
|
|
||||||
|
## Comparison
|
||||||
|
|
||||||
|
The following table lists estimated per-unit characteristics. All numbers are unvalidated estimates pending synthesis. Relative area is normalized to A1 (1×); the actual ratios depend on the technology library, target frequency, and choice of final adder. For B7 the per-cluster area is given; the per-core effective area is per-cluster area divided by 8.
|
||||||
|
|
||||||
|
| Design | MUL Latency | MUL Throughput | DIV Latency | DIV Throughput | Relative Area (per unit) | Per-core effective area | Verification Complexity |
|
||||||
|
|-------------------------------------------------|-------------|----------------|-------------|----------------|----------------------------------|-------------------------|-------------------------|
|
||||||
|
| A1. Shift-add + restoring | 64 | 1/64 | 64 | 1/64 | 1× | 1× | Low |
|
||||||
|
| A2. Booth-r4 + NR div | 2 | 1/1 | 64 | 1/64 | ~3–4× (unvalidated) | ~3–4× | Low–Medium |
|
||||||
|
| A3. Pipelined r4/r8 + SRT-4 (non-pipe) | 2–3 | 1/1 | 16 | 1/16 | ~5–7× (unvalidated) | ~5–7× | Medium |
|
||||||
|
| A4. Combinational + SRT-16 | 1 | 1/1 | 4–8 | 1/4–1/8 | ~10–15× (unvalidated) | ~10–15× | High |
|
||||||
|
| B1. 2-stage r4 + NR-64 (no early-exit) | 2 | 1/1 | 64 worst | 1/64 | ~3–5× (unvalidated) | ~3–5× | Low–Medium |
|
||||||
|
| B2. 3-stage r8 + NR-64 | 3 | 1/1 | 64 worst | 1/64 | ~4–6× (unvalidated) | ~4–6× | Medium |
|
||||||
|
| B4. SRT-4 (non-pipelined), divider only | n/a | n/a | 16 | 1/16 | adds ~3–4× to A2 divider (unval.) | n/a (divider) | Medium |
|
||||||
|
| B4-pipe. SRT-4 (pipelined, 1/cycle), divider only| n/a | n/a | 16 | 1/1 | adds ~6–8× to A2 total (unval.) | n/a (divider) | Medium–High |
|
||||||
|
| B5. Newton–Raphson (64-bit) | 1 | 1/1 | 6–10 | 1/6–1/10 | ~6–8× (unvalidated) | ~6–8× | High |
|
||||||
|
| B6. Combinational non-restoring array divider | 1 | 1/1 | 1 | 1/1 | very large (unvalidated) | very large | Medium |
|
||||||
|
| B7. Cluster-shared (per 8 cores) B1 + B4-pipe | 2 | 1/1 | 16 | 1/1 (when free)| per-cluster ~8–12× A1 (unval.) | ~1–1.5× A1 per core | Medium–High |
|
||||||
|
|
||||||
|
The previous revision compared only per-core designs and omitted the per-core effective area column for B7. The per-core effective area for B7 is approximately per-cluster area divided by 8, which makes the cluster organization attractive only if the per-core MUL/DIV unit would otherwise exceed the per-core area budget by a factor of 5–8× or more. The trade-off (per-core area savings vs. inter-core contention latency) is not quantified in this document.
|
||||||
|
|
||||||
|
## Proposal (Resolved, Conditional on Synthesis)
|
||||||
|
|
||||||
|
PROPOSAL (PRIMARY, CONDITIONAL): Adopt B1 (2-stage pipelined Radix-4 Booth multiplier, 33 partial products, 4:2 compressor tree, 2-cycle latency, 1/cycle throughput) paired with a 64-cycle radix-2 non-restoring iterative divider with explicit worst-case latency. The 32-cycle figure in the previous revision is corrected to 64.
|
||||||
|
|
||||||
|
CONDITIONAL UPGRADE: If workload analysis (TBD) or synthesis results show that 64-cycle divide latency is a bottleneck, upgrade the divider to B4-pipe (pipelined radix-4 SRT, 16-cycle latency, 1/cycle throughput). The trigger condition is empirical and is not assumed to hold by default.
|
||||||
|
|
||||||
|
ORGANIZATIONAL FALLBACK: If per-core area constraints prove tighter than current unvalidated estimates, evaluate B7 (cluster-shared organization, 16 instances serving 8 cores each, FIFO interface) with a per-cluster B1 + B4-pipe datapath. The trigger condition is a per-core area budget exceeded by B1's estimated footprint.
|
||||||
|
|
||||||
|
This resolves the prior contradiction between the abstract PROPOSAL (which named a pipelined SRT-4) and the final Recommendation (which named a 32-cycle non-restoring divider). The remaining sections are aligned to B1 + 64-cycle non-restoring divider as the primary proposal, with B4-pipe as a conditional upgrade and B7 as a separate organizational fallback. The proposal is conditional on synthesis data and on a per-core area budget that has not yet been established.
|
||||||
|
|
||||||
|
## Advantages
|
||||||
|
|
||||||
|
- 2-cycle pipelined Radix-4 Booth multiplier offers 1/cycle throughput at modest per-core area, suitable for replicated 128-core operation.
|
||||||
|
- 64-cycle non-restoring iterative divider has explicit worst-case latency that does not depend on operand values, simplifying pipeline scheduling, forwarding, and verification.
|
||||||
|
- Radix-4 Booth and radix-2 non-restoring division are well-understood and have reference implementations in Rocket, BOOM, and Ariane (specific measurements: INSUFFICIENT EVIDENCE in the XH-1 repository; see Sources).
|
||||||
|
|
||||||
|
## Disadvantages
|
||||||
|
|
||||||
|
- 64-cycle divide is slow for workloads dominated by large-integer arithmetic (RSA, big-integer math, certain cryptographic primitives). The conditional B4-pipe upgrade addresses this at additional area cost.
|
||||||
|
- A non-restoring iterative divider has the lowest per-core area but penalizes every divide by up to 64 cycles, which can be felt in hash-table probing, parser/lexer dispatch, and some interpreter dispatch loops.
|
||||||
|
- Booth encoding complicates verification of signed/unsigned correctness for MULH / MULHSU / MULHU; the verification plan must cover all four sign combinations explicitly.
|
||||||
|
- The design's performance on divide-heavy workloads depends on the in-order / out-of-order issue model, which is TBD.
|
||||||
|
|
||||||
|
## XH-1 Considerations
|
||||||
|
|
||||||
|
- ASSUMPTION: XH-1 is a 128-core design with a short pipeline (TBD by `pipeline.md`). The MUL/DIV unit must fit in a small per-core area budget and the EX-stage latency budget.
|
||||||
|
- ASSUMPTION: The design targets a balance of general-purpose and HPC-adjacent workloads; therefore a moderately fast (but not the fastest) divider is acceptable as the default, with an explicit upgrade path.
|
||||||
|
- The multiplier is sized to produce the full 128-bit product so that MULH/MULHU selection requires only output muxing, not a separate datapath.
|
||||||
|
- The W-variants (MULW, DIVW, DIVUW, REMW, REMUW) reuse the lower 32 bits of the 64-bit datapath with sign-extension at the output, not a separate 32-bit datapath.
|
||||||
|
|
||||||
|
### Instruction Coverage
|
||||||
|
|
||||||
|
- MUL/MULH/MULHSU/MULHU: supported.
|
||||||
|
- DIV/DIVU/REM/REMU: supported, with RISC-V-spec corner cases handled explicitly.
|
||||||
|
- MULW/DIVW/DIVUW/REMW/REMUW: supported via shared datapath.
|
||||||
|
|
||||||
|
### Edge Cases (RISC-V Spec, Document Version 20191213, Chapter 7)
|
||||||
|
|
||||||
|
The following are the architecturally specified results. The previous revision contained an error in the DIVU-by-zero description; the corrected behavior, stated consistently using two's-complement bit patterns, is:
|
||||||
|
|
||||||
|
- DIV by zero: quotient = 2^XLEN − 1 (all bits set, which is the two's-complement representation of −1); remainder = dividend (x).
|
||||||
|
- DIVU by zero: quotient = 2^XLEN − 1 (all bits set, which is also the two's-complement representation of −1 for an XLEN-bit signed interpretation, but is the unsigned all-ones value); remainder = dividend (x).
|
||||||
|
- REM by zero: remainder = dividend (x); quotient = 2^XLEN − 1 (two's-complement −1).
|
||||||
|
- REMU by zero: remainder = dividend (x); quotient = 2^XLEN − 1 (unsigned all-ones).
|
||||||
|
- Signed overflow (DIV of INT64_MIN by −1): quotient = INT64_MIN, remainder = 0.
|
||||||
|
- REM sign rule: the sign of the remainder follows the sign of the dividend. This is a frequent bug source and must be covered explicitly in verification; the previous revision did not call it out.
|
||||||
|
|
||||||
|
Note on terminology: "quotient = −1" and "quotient = 2^XLEN − 1" describe the same bit pattern in two's complement. This document uses the unsigned 2^XLEN − 1 form throughout to avoid ambiguity about sign interpretation. The unit must produce these results without raising an exception.
|
||||||
|
|
||||||
|
## 128-Core Scalability
|
||||||
|
|
||||||
|
- Per-core area: a small 64×64 → 128-bit Radix-4 Booth multiplier with a 4:2 compressor tree, a 2-stage pipelined final adder, and a 64-cycle non-restoring divider is expected to be a small fraction of a typical RV core area, but the exact fraction is INSUFFICIENT EVIDENCE because no baseline core area has been established for XH-1.
|
||||||
|
- Floorplanning: a regular MUL/DIV layout that mirrors across all 128 cores is preferred to avoid routing asymmetry that would break clock distribution and thermal symmetry.
|
||||||
|
- Voltage / frequency: a moderately pipelined MUL/DIV is more resilient to voltage droop; this is an advantage for 128-core operation.
|
||||||
|
- Contention: per-core replication means there is no inter-core contention for the MUL/DIV unit. The only contention is intra-core (e.g., two dependent divides in flight in an out-of-order pipeline). The cluster-shared B7 organization reintroduces inter-core contention; this is the central trade-off.
|
||||||
|
- Test / DFT: 128 instances of the MUL/DIV unit (or 16 instances in B7) must be tested. A scan-friendly, fully synchronous design with no asynchronous reset paths inside the iterative divider is preferable. DFT strategy is addressed in a dedicated section below.
|
||||||
|
|
||||||
|
## Performance Considerations
|
||||||
|
|
||||||
|
- For general-purpose code, MUL/DIV is rarely the bottleneck; a 2-cycle MUL latency matches typical issue-to-use distances.
|
||||||
|
- For cryptography, the relevant metric depends on the algorithm. Karatsuba multiplication and certain Montgomery multiplication formulations (e.g., CIOS, FIOS) use full 64×64→128 multiplies and select either the lower or upper half depending on the step; MULH throughput is therefore relevant to some Montgomery and Karatsuba sequences, not only to software-emulated 128-bit integers. The previous revision's claim that MULH is irrelevant to Montgomery/Karatsuba is oversimplified and is corrected here to a softer form: MULH matters for software-emulated 128-bit integers and for some Montgomery / Karatsuba formulations; the precise relevance is algorithm-dependent. A specific algorithm study is out of scope for this document.
|
||||||
|
- For hash tables and interpreters, DIV latency matters more than throughput; a 64-cycle divider is acceptable but not ideal.
|
||||||
|
- For 32-bit integer code, the W-variants are the hot path; reusing the 64-bit datapath with muxed operands is acceptable.
|
||||||
|
- A 64-cycle divide in a deeply pipelined in-order core can be tolerated if the divider is non-blocking and the result is forwarded late; the claim that "in-order cannot tolerate a 64-cycle divider" is too strong and is not made here.
|
||||||
|
|
||||||
|
## Area Considerations
|
||||||
|
|
||||||
|
- ASSUMPTION (unvalidated, heuristic): A 64×64 → 128-bit Radix-4 Booth multiplier with a 4:2 compressor tree and 2-stage pipelined final adder is on the order of 0.02–0.10 mm² in a typical 7nm process, depending on target frequency, Vdd, and FF corner. No source is cited; INSUFFICIENT EVIDENCE to refine this without a technology file. The 10× range is not a die-budget input and must be replaced with synthesis data.
|
||||||
|
- ASSUMPTION (unvalidated, heuristic): A 64-cycle non-restoring iterative divider is on the order of 0.01–0.05 mm² in the same envelope.
|
||||||
|
- PROPOSAL (unvalidated): Total MUL/DIV area target: <0.15 mm² per core. 128× replication: ~20 mm². These numbers are order-of-magnitude estimates and must be replaced with synthesis data before being used for die budgeting. The previous revision provided a similar target without a baseline; the same caveat applies.
|
||||||
|
- The choice between a Wallace and a Dadda tree implemented with 4:2 compressors is largely a layout / regularity preference; INSUFFICIENT EVIDENCE to prefer one over the other without synthesis. The claim that the 4:2 compressor tree is "more area-efficient than a Wallace tree" is removed.
|
||||||
|
- Sign-extension and zero-extension muxes for MULW are negligible area.
|
||||||
|
|
||||||
|
## Power and Energy Considerations
|
||||||
|
|
||||||
|
- A 64×64 multiplier tree has high switching activity; clock gating when the unit is idle is essential.
|
||||||
|
- The iterative divider has lower average switching power than a fully combinational divider, but its long residency increases leakage energy per operation.
|
||||||
|
- Power gating: at 128 cores, a per-core power-gate for the MUL/DIV unit is worth considering if idle periods dominate. The wake-up latency and IR-drop impact on the power grid are not yet analyzed; INSUFFICIENT EVIDENCE without a full-die power analysis.
|
||||||
|
- Energy per multiply is heuristic: energy tends to scale with the number of switching nodes in the critical reduction tree. This is a rule of thumb, not a measured result; INSUFFICIENT EVIDENCE for a specific quantitative claim.
|
||||||
|
- Energy per divide is dominated by the 64-cycle residency; data-dependent early-exit (if implemented) reduces energy for typical operands but the worst-case energy remains.
|
||||||
|
|
||||||
|
## Implementation Considerations
|
||||||
|
|
||||||
|
- Synchronous, single-clock-domain design inside the unit.
|
||||||
|
- No asynchronous resets inside the iterative divider; synchronous reset only at the start of an operation.
|
||||||
|
- Final carry-propagate adder: Kogge–Stone, Han–Carlson, and Brent–Kung are all viable. The choice depends on the EX-stage timing budget and the area target; Kogge–Stone / Han–Carlson are faster but larger, Brent–Kung is smaller but slower. The previous revision named a default without justification; this document records the choice as a trade-off driven by the EX-stage timing budget, to be decided after synthesis.
|
||||||
|
- Divider quotient and remainder registers are 64 bits each, with an extra bit for the iterative sign.
|
||||||
|
- The microarchitectural state machine is small and well-suited to a one-hot or binary-encoded FSM.
|
||||||
|
- Output muxing for MUL / MULH / MULHSU / MULHU / MULW is a small mux tree, not a separate datapath.
|
||||||
|
- Booth encoding produces 33 partial products for radix-4 of a 64-bit operand (32 signed-digit rows plus one sign-correction row). The sign-correction row is required for negative-operand correctness; verification must cover it explicitly.
|
||||||
|
|
||||||
|
## Verification Considerations
|
||||||
|
|
||||||
|
- Formal verification of the multiplier compressor tree and final adder is feasible with bounded model checkers and is recommended.
|
||||||
|
- Directed tests for division edge cases: ÷0 (signed and unsigned, both quotient and remainder), INT64_MIN / −1, dividend = divisor, dividend = 0, divisor = 1, all-ones, alternating bits, dividend = −1, divisor = 2.
|
||||||
|
- Explicit coverage of the REM sign-of-dividend rule.
|
||||||
|
- Coverage of all 4 MUL variants (MUL, MULH, MULHSU, MULHU) and the W-variants.
|
||||||
|
- Randomized differential testing against a software reference (a GCC-compiled test harness running on Spike or QEMU, or a Python / C++ golden model) is recommended.
|
||||||
|
- 128-core DFT: see the dedicated section below.
|
||||||
|
|
||||||
|
### DFT Strategy (128 Replicated Units)
|
||||||
|
|
||||||
|
- Single-clock-domain, synchronous-reset-only design is required for scan insertion.
|
||||||
|
- Each MUL/DIV instance is scan-stitched independently; long scan chains between the iterative divider and surrounding logic are avoided to prevent hold-time issues.
|
||||||
|
- Scan compression: per-core compression reduces the number of top-level scan pins; the compression architecture is TBD by the DFT plan.
|
||||||
|
- BIST: optional per-core BIST for the MUL/DIV unit is feasible given its small size and regular structure; this would reduce ATPG complexity at the cost of additional area for the BIST controller.
|
||||||
|
- ATPG implications: the iterative divider is the only sequential element of consequence in the MUL/DIV block; full-scan coverage is straightforward if no asynchronous paths are introduced.
|
||||||
|
|
||||||
|
## Software Implications
|
||||||
|
|
||||||
|
- Compilers emit MUL freely; no software changes are required.
|
||||||
|
- Division by a constant is often transformed by the compiler into a magic-number multiply; the choice of MUL/DIV design therefore disproportionately affects runtime divide performance for code with frequent constant divides.
|
||||||
|
- For languages with software-emulated 128-bit integers (`__int128` in C/C++), the compiler emits MULH / MULHU sequences; MULH latency and throughput are the relevant metrics here. This is the primary case where MULH is the key metric.
|
||||||
|
- Crypto libraries (libsodium, OpenSSL, mbedTLS) use Karatsuba and Montgomery multiplication formulations whose reliance on MULH vs. MUL is algorithm-dependent; MULH throughput is relevant to some of these formulations and not to others. The previous revision's strong claim that MULH is irrelevant to Karatsuba / Montgomery is corrected to a softer, algorithm-dependent statement.
|
||||||
|
- JavaScript engines and language runtimes may issue frequent DIVs for tagged-value unpacking; a 64-cycle divider increases interpreter dispatch latency for that pattern.
|
||||||
|
|
||||||
|
## Recommendation
|
||||||
|
|
||||||
|
RECOMMENDATION (CONDITIONAL ON SYNTHESIS): Subject to validation against a target technology library, target frequency, and a per-core area budget, adopt B1 (2-stage pipelined Radix-4 Booth multiplier, 33 partial products, 4:2 compressor tree, 2-cycle latency, 1/cycle throughput) paired with a 64-cycle radix-2 non-restoring iterative divider with explicit worst-case latency and no data-dependent early-exit in the base configuration.
|
||||||
|
|
||||||
|
Rationale:
|
||||||
|
|
||||||
|
- Best balance of area, energy, and performance for 128-core replication under current unvalidated estimates, pending synthesis.
|
||||||
|
- Deterministic, fully synchronous design simplifies verification and DFT.
|
||||||
|
- 1/cycle MUL throughput meets general-purpose and crypto lower-half-multiply needs.
|
||||||
|
- 64-cycle worst-case DIV latency is acceptable for general-purpose workloads; an upgrade path to B4-pipe (16-cycle pipelined SRT-4) is reserved for the case where profiling evidence supports it.
|
||||||
|
|
||||||
|
The recommendation is conditional because all area, energy, and latency numbers in this document are unvalidated estimates; the specific B1 + 64-cycle non-restoring choice depends on a per-core area budget that has not yet been established. If synthesis shows that B1 exceeds the per-core area budget, B7 (cluster-shared organization) is the next design point to evaluate before reducing the per-core divider latency.
|
||||||
|
|
||||||
|
RECOMMENDATION (CONDITIONAL UPGRADE): If early workload analysis (TBD) or profiling evidence shows that divide latency is a bottleneck, upgrade the divider to B4-pipe (pipelined radix-4 SRT, 16-cycle latency, 1/cycle throughput) at an estimated additional ~3–4× divider area on top of B1's divider (unvalidated). Do not adopt a Newton–Raphson divider unless profiling evidence strongly supports it, due to verification complexity and variable latency.
|
||||||
|
|
||||||
|
RECOMMENDATION: Do not adopt a fully combinational non-restoring array divider or a fully combinational multiplier for the replicated 128-core design. The area cost is not justified by the latency benefit at the per-core replication factor.
|
||||||
|
|
||||||
|
RECOMMENDATION (FALLBACK): If per-core area constraints prove tighter than estimated, evaluate B7 (cluster-shared organization, 16 instances serving 8 cores each, FIFO interface) with a per-cluster B1 + B4-pipe datapath before reducing the per-core divider latency further. This trades inter-core contention for per-core area and is the right knob to pull when the per-core area budget is the binding constraint.
|
||||||
|
|
||||||
|
## Confidence
|
||||||
|
|
||||||
|
- Direction of recommendation: MEDIUM. The general design class is well-established; the specific B1 + 64-cycle non-restoring choice depends on a per-core area budget and process node that have not yet been validated.
|
||||||
|
- Quantitative area, latency, power, and energy numbers: LOW. No measurements exist in the XH-1 repository; all such numbers in this document are explicitly labeled as unvalidated estimates or INSUFFICIENT EVIDENCE.
|
||||||
|
- Verification strategy approach: MEDIUM. The recommended approach (formal on the multiplier, directed + randomized for the divider, explicit REM sign rule) is standard practice. Whether the XH-1 implementation passes the verification plan is not yet known.
|
||||||
|
- Spec-level edge-case correctness (i.e., that the RISC-V spec defines the behavior as documented here): MEDIUM. The RISC-V Unprivileged ISA, Document Version 20191213, Chapter 7 is cited as the source; whether the XH-1 implementation matches the spec is not yet verified and is the subject of the verification plan, not a research claim.
|
||||||
|
|
||||||
|
## Open Questions
|
||||||
|
|
||||||
|
- What is the EX-stage latency budget? (Depends on `pipeline.md`.)
|
||||||
|
- What is the target frequency and process node? Determines whether 2-cycle MUL is feasible, and which final-adder architecture is appropriate.
|
||||||
|
- Is the design in-order or out-of-order? Out-of-order execution can hide divide latency; in-order can also tolerate a 64-cycle divider if the divider is non-blocking and the result is forwarded late, but the scheduling cost depends on the specific pipeline depth and issue model.
|
||||||
|
- Will the MUL/DIV unit share an issue port with the ALU, or have a dedicated issue port? The recommendation assumes the MUL/DIV unit has access to the issue port, but whether the port is shared or dedicated is TBD. If the port is shared with the ALU, the per-cycle issue bandwidth of the MUL/DIV unit must be reconciled with the ALU's, and the throughput numbers in the comparison table must be reinterpreted as "throughput when the issue port is available." A shared port may force B1 throughput below 1/cycle on divide-heavy code.
|
||||||
|
- Is there a future F / D extension? If so, the integer MUL/DIV unit may also need to feed FP-to-int conversions or FP reciprocal iterations; this is not yet analyzed.
|
||||||
|
- What is the expected workload mix? General-purpose vs. HPC vs. embedded vs. server?
|
||||||
|
- Should the MUL/DIV unit be power-gated when idle? At 128 cores, idle probability may be high; wake-up latency and IR-drop impact are TBD.
|
||||||
|
- Should the divider support a "fast-path" for division by a small constant (e.g., a compiler-inserted reciprocal-multiply hint) as a microarchitectural feature? This is recorded as a TBD feature and is not assumed in the base configuration.
|
||||||
|
- Is the B7 cluster-shared organization a realistic fallback, or is per-core replication a hard requirement?
|
||||||
|
- What scan-compression architecture will be used for 128 replicated units?
|
||||||
|
|
||||||
|
## Sources
|
||||||
|
|
||||||
|
- RISC-V Unprivileged ISA, Document Version 20191213, Chapter 7, "M Extension." Defines MUL, MULH, MULHSU, MULHU, DIV, DIVU, REM, REMU, and the W-variants, including the corner-case behavior for division by zero and signed overflow. The "−1" and "2^XLEN − 1" quotient values for ÷0 are the same bit pattern in two's complement and are documented in this revision of the spec.
|
||||||
|
- Rocket Chip, file `src/main/scala/rocket/MulDiv.scala`, public repository. Background reference for typical MUL/DIV organization and reported latency ranges. Specific commit, measured numbers, and PPA data: INSUFFICIENT EVIDENCE in the XH-1 repository to cite.
|
||||||
|
- BOOM (SmallBoomConfig and MediumBoomConfig), public repository. Background reference only.
|
||||||
|
- Ariane, file `core/multiplier.sv` and related, public repository. Background reference only.
|
||||||
|
- Hennessy and Patterson, "Computer Architecture: A Quantitative Approach." General MUL/DIV trade-off discussion. Specific chapter / edition: INSUFFICIENT EVIDENCE in the XH-1 repository to cite.
|
||||||
|
- Parhami, "Computer Arithmetic: Algorithms and Hardware Designs." Standard reference for multiplier and divider algorithms including Booth, Wallace / Dadda, SRT, and Newton–Raphson. Specific chapter / edition: INSUFFICIENT EVIDENCE in the XH-1 repository to cite.
|
||||||
|
- No quantitative claim in this document is derived from a measurement of XH-1 silicon, layout, or synthesis. All such numbers are explicitly labeled as unvalidated estimates, heuristics, or INSUFFICIENT EVIDENCE.
|
||||||
+307
File diff suppressed because one or more lines are too long
+525
-608
File diff suppressed because it is too large
Load Diff
Executable
+287
@@ -0,0 +1,287 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
# ============================================================
|
||||||
|
# XH-1 Research Harness
|
||||||
|
# Autonomous Research / Review / Revision Pipeline
|
||||||
|
#
|
||||||
|
# Pure Bash
|
||||||
|
# OpenAI-compatible APIs
|
||||||
|
# Designed for OpenRouter
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
ROOT="${XH1_RESEARCH_ROOT:-research}"
|
||||||
|
CONFIG="${XH1_RESEARCH_CONFIG:-xh1-research.conf}"
|
||||||
|
STATE="${ROOT}/.xh1"
|
||||||
|
|
||||||
|
mkdir -p \
|
||||||
|
"$STATE/logs" \
|
||||||
|
"$STATE/responses" \
|
||||||
|
"$STATE/runs"
|
||||||
|
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
# Defaults
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
|
||||||
|
API_BASE_URL="${API_BASE_URL:-https://openrouter.ai/api/v1}"
|
||||||
|
API_KEY_ENV="${API_KEY_ENV:-OPENROUTER_API_KEY}"
|
||||||
|
|
||||||
|
RESEARCH_MODEL="${RESEARCH_MODEL:-}"
|
||||||
|
REVIEW_MODEL="${REVIEW_MODEL:-$RESEARCH_MODEL}"
|
||||||
|
REVISION_MODEL="${REVISION_MODEL:-$RESEARCH_MODEL}"
|
||||||
|
|
||||||
|
MAX_RESEARCH_ROUNDS="${MAX_RESEARCH_ROUNDS:-3}"
|
||||||
|
MAX_API_RETRIES="${MAX_API_RETRIES:-3}"
|
||||||
|
|
||||||
|
DELAY_SECONDS="${DELAY_SECONDS:-5}"
|
||||||
|
MAX_ITERATIONS="${MAX_ITERATIONS:-0}"
|
||||||
|
|
||||||
|
REVIEW_ENABLED="${REVIEW_ENABLED:-true}"
|
||||||
|
AUTO_COMMIT="${AUTO_COMMIT:-false}"
|
||||||
|
|
||||||
|
RUN_ID=""
|
||||||
|
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
# General
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
|
||||||
|
die() {
|
||||||
|
echo "ERROR: $*" >&2
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
|
||||||
|
log() {
|
||||||
|
local message="$*"
|
||||||
|
|
||||||
|
echo "[$(date -u '+%Y-%m-%dT%H:%M:%SZ')] $message" \
|
||||||
|
| tee -a "$STATE/logs/harness.log"
|
||||||
|
}
|
||||||
|
|
||||||
|
require() {
|
||||||
|
command -v "$1" >/dev/null 2>&1 ||
|
||||||
|
die "Required command not found: $1"
|
||||||
|
}
|
||||||
|
|
||||||
|
require_dependencies() {
|
||||||
|
require bash
|
||||||
|
require curl
|
||||||
|
require jq
|
||||||
|
require find
|
||||||
|
require grep
|
||||||
|
require sed
|
||||||
|
require awk
|
||||||
|
}
|
||||||
|
|
||||||
|
timestamp() {
|
||||||
|
date -u '+%Y%m%dT%H%M%SZ'
|
||||||
|
}
|
||||||
|
|
||||||
|
load_config() {
|
||||||
|
if [[ -f "$CONFIG" ]]; then
|
||||||
|
# shellcheck disable=SC1090
|
||||||
|
source "$CONFIG"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
get_api_key() {
|
||||||
|
local key="${!API_KEY_ENV:-}"
|
||||||
|
|
||||||
|
[[ -n "$key" ]] ||
|
||||||
|
die "Missing API key. Set ${API_KEY_ENV}."
|
||||||
|
|
||||||
|
printf '%s' "$key"
|
||||||
|
}
|
||||||
|
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
# Configuration
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
|
||||||
|
init_config() {
|
||||||
|
|
||||||
|
if [[ -f "$CONFIG" ]]; then
|
||||||
|
echo "Configuration already exists:"
|
||||||
|
echo " $CONFIG"
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
|
||||||
|
cat > "$CONFIG" <<'CONFIG'
|
||||||
|
# ============================================================
|
||||||
|
# XH-1 Research Harness Configuration
|
||||||
|
# ============================================================
|
||||||
|
|
||||||
|
# OpenAI-compatible API endpoint.
|
||||||
|
# OpenRouter:
|
||||||
|
API_BASE_URL="https://openrouter.ai/api/v1"
|
||||||
|
|
||||||
|
# Environment variable containing the API key.
|
||||||
|
API_KEY_ENV="OPENROUTER_API_KEY"
|
||||||
|
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
# Models
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
|
||||||
|
# Model used for initial research.
|
||||||
|
RESEARCH_MODEL=""
|
||||||
|
|
||||||
|
# Model used for independent review.
|
||||||
|
REVIEW_MODEL=""
|
||||||
|
|
||||||
|
# Model used to revise rejected research.
|
||||||
|
REVISION_MODEL=""
|
||||||
|
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
# Research behavior
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
|
||||||
|
# Maximum Research -> Review -> Revision cycles for ONE topic.
|
||||||
|
MAX_RESEARCH_ROUNDS=3
|
||||||
|
|
||||||
|
# API retries per individual request.
|
||||||
|
MAX_API_RETRIES=3
|
||||||
|
|
||||||
|
# Delay between topics.
|
||||||
|
DELAY_SECONDS=5
|
||||||
|
|
||||||
|
# 0 = unlimited.
|
||||||
|
MAX_ITERATIONS=0
|
||||||
|
|
||||||
|
# Enable independent reviewer.
|
||||||
|
REVIEW_ENABLED=true
|
||||||
|
|
||||||
|
# Automatically create git commits.
|
||||||
|
AUTO_COMMIT=false
|
||||||
|
CONFIG
|
||||||
|
|
||||||
|
echo "Created $CONFIG"
|
||||||
|
}
|
||||||
|
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
# Run state
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
|
||||||
|
start_run() {
|
||||||
|
|
||||||
|
RUN_ID="$(timestamp)"
|
||||||
|
|
||||||
|
mkdir -p "$STATE/runs/$RUN_ID"
|
||||||
|
|
||||||
|
: > "$STATE/runs/$RUN_ID/completed"
|
||||||
|
: > "$STATE/runs/$RUN_ID/failed"
|
||||||
|
: > "$STATE/runs/$RUN_ID/attempts"
|
||||||
|
|
||||||
|
log "Started run: $RUN_ID"
|
||||||
|
}
|
||||||
|
|
||||||
|
run_has_completed() {
|
||||||
|
local file="$1"
|
||||||
|
|
||||||
|
grep -Fxq "$file" \
|
||||||
|
"$STATE/runs/$RUN_ID/completed" 2>/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
run_has_failed() {
|
||||||
|
local file="$1"
|
||||||
|
|
||||||
|
grep -Fxq "$file" \
|
||||||
|
"$STATE/runs/$RUN_ID/failed" 2>/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
mark_completed() {
|
||||||
|
local file="$1"
|
||||||
|
|
||||||
|
printf '%s\n' "$file" \
|
||||||
|
>> "$STATE/runs/$RUN_ID/completed"
|
||||||
|
}
|
||||||
|
|
||||||
|
mark_failed() {
|
||||||
|
local file="$1"
|
||||||
|
|
||||||
|
printf '%s\n' "$file" \
|
||||||
|
>> "$STATE/runs/$RUN_ID/failed"
|
||||||
|
}
|
||||||
|
|
||||||
|
record_attempt() {
|
||||||
|
|
||||||
|
local file="$1"
|
||||||
|
local round="$2"
|
||||||
|
local phase="$3"
|
||||||
|
local result="$4"
|
||||||
|
|
||||||
|
printf '%s\t%s\t%s\t%s\t%s\n' \
|
||||||
|
"$(date -u '+%Y-%m-%dT%H:%M:%SZ')" \
|
||||||
|
"$file" \
|
||||||
|
"$round" \
|
||||||
|
"$phase" \
|
||||||
|
"$result" \
|
||||||
|
>> "$STATE/runs/$RUN_ID/attempts"
|
||||||
|
}
|
||||||
|
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
# Research discovery
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
|
||||||
|
is_soon() {
|
||||||
|
grep -qE '^SOON[[:space:]]*$' "$1"
|
||||||
|
}
|
||||||
|
|
||||||
|
find_next_task() {
|
||||||
|
|
||||||
|
while IFS= read -r -d '' file; do
|
||||||
|
|
||||||
|
[[ "$file" == "$STATE"/* ]] && continue
|
||||||
|
|
||||||
|
if is_soon "$file"; then
|
||||||
|
|
||||||
|
if ! run_has_completed "$file" &&
|
||||||
|
! run_has_failed "$file"
|
||||||
|
then
|
||||||
|
printf '%s\n' "$file"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
fi
|
||||||
|
|
||||||
|
done < <(
|
||||||
|
find "$ROOT" \
|
||||||
|
-type f \
|
||||||
|
-name '*.md' \
|
||||||
|
-not -path "$STATE/*" \
|
||||||
|
-print0
|
||||||
|
)
|
||||||
|
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
# Context
|
||||||
|
# ------------------------------------------------------------
|
||||||
|
|
||||||
|
build_context() {
|
||||||
|
|
||||||
|
local file="$1"
|
||||||
|
|
||||||
|
local relative="${file#"$ROOT"/}"
|
||||||
|
local area="${relative%%/*}"
|
||||||
|
|
||||||
|
cat <<EOF
|
||||||
|
XH-1 RESEARCH PROJECT
|
||||||
|
=====================
|
||||||
|
|
||||||
|
Project:
|
||||||
|
XH-1
|
||||||
|
|
||||||
|
Architecture:
|
||||||
|
Custom 128-core RISC-V processor
|
||||||
|
|
||||||
|
Research repository:
|
||||||
|
XH-1 Research
|
||||||
|
|
||||||
|
Current document:
|
||||||
|
$file
|
||||||
|
|
||||||
|
Research area:
|
||||||
|
$area
|
||||||
|
|
||||||
|
CURRENT DOCUMENT
|
||||||
|
================
|
||||||
|
|
||||||
+8
-3
@@ -13,16 +13,21 @@
|
|||||||
# Other OpenAI-compatible providers:
|
# Other OpenAI-compatible providers:
|
||||||
# https://example.com/v1
|
# https://example.com/v1
|
||||||
|
|
||||||
API_BASE_URL="https://api.openai.com/v1"
|
API_BASE_URL="https://openrouter.ai/api/v1"
|
||||||
|
|
||||||
# Environment variable containing your API key.
|
# Environment variable containing your API key.
|
||||||
API_KEY_ENV="OPENAI_API_KEY"
|
API_KEY_ENV="OPENROUTER_API_KEY"
|
||||||
|
|
||||||
|
|
||||||
# Model to use.
|
# Model to use.
|
||||||
MODEL="YOUR_MODEL_HERE"
|
MODEL="minimax/minimax-m3:free"
|
||||||
|
RESEARCH_MODEL="minimax/minimax-m3:free"
|
||||||
|
REVIEW_MODEL="minimax/minimax-m3:free"
|
||||||
|
REVISION_MODEL="minimax/minimax-m3:free"
|
||||||
|
|
||||||
# Retry failed API calls.
|
# Retry failed API calls.
|
||||||
MAX_RETRIES=3
|
MAX_RETRIES=3
|
||||||
|
MAX_RESEARCH_ROUNDS=4
|
||||||
|
|
||||||
# Delay between research tasks.
|
# Delay between research tasks.
|
||||||
DELAY_SECONDS=5
|
DELAY_SECONDS=5
|
||||||
|
|||||||
Reference in New Issue
Block a user