feat: publish FreeRTOS C FC06 card

This commit is contained in:
2026-07-19 16:36:03 +02:00
commit 0fc6158501
195 changed files with 60591 additions and 0 deletions
+269
View File
@@ -0,0 +1,269 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
`default_nettype none
module hazard3_alu #(
`include "hazard3_config.vh"
,
`include "hazard3_width_const.vh"
) (
input wire [W_ALUOP-1:0] aluop,
input wire [6:0] funct7_32b,
input wire [2:0] funct3_32b,
input wire [W_DATA-1:0] op_a,
input wire [W_DATA-1:0] op_b,
output reg [W_DATA-1:0] result,
output wire cmp
);
`include "hazard3_ops.vh"
// ----------------------------------------------------------------------------
// Fiddle around with add/sub, comparisons etc (all related).
wire sub = !(aluop == ALUOP_ADD || (|EXTENSION_ZBA && aluop == ALUOP_SHXADD));
wire inv_op_b = sub && !(
aluop == ALUOP_AND || aluop == ALUOP_OR || aluop == ALUOP_XOR || aluop == ALUOP_RS2
);
wire [W_DATA-1:0] op_a_shifted =
|EXTENSION_ZBA && aluop == ALUOP_SHXADD ? (
!funct3_32b[2] ? op_a << 1 :
!funct3_32b[1] ? op_a << 2 : op_a << 3
) : op_a;
wire [W_DATA-1:0] op_b_inv = op_b ^ {W_DATA{inv_op_b}};
wire [W_DATA-1:0] sum = op_a_shifted + op_b_inv + {{W_DATA-1{1'b0}}, sub};
wire [W_DATA-1:0] op_xor = op_a ^ op_b;
wire cmp_is_unsigned = aluop == ALUOP_LTU ||
|EXTENSION_ZBB && aluop == ALUOP_MAXU ||
|EXTENSION_ZBB && aluop == ALUOP_MINU;
wire lt = op_a[W_DATA-1] == op_b[W_DATA-1] ? sum[W_DATA-1] :
cmp_is_unsigned ? op_b[W_DATA-1] : op_a[W_DATA-1] ;
assign cmp = aluop == ALUOP_SUB ? |op_xor : lt;
// ----------------------------------------------------------------------------
// Separate units for shift, ctz etc
wire [W_DATA-1:0] shift_dout;
wire shift_right_nleft =
aluop == ALUOP_SRL ||
aluop == ALUOP_SRA ||
(|EXTENSION_ZBB && aluop == ALUOP_ROR ) ||
(|EXTENSION_ZBS && aluop == ALUOP_BEXT ) ||
(|EXTENSION_XH3BEXTM && aluop == ALUOP_BEXTM);
wire shift_arith = aluop == ALUOP_SRA;
wire shift_rotate = |EXTENSION_ZBB & (aluop == ALUOP_ROR || aluop == ALUOP_ROL);
hazard3_shift_barrel #(
`include "hazard3_config_inst.vh"
) shifter (
.din (op_a),
.shamt (op_b[4:0]),
.right_nleft (shift_right_nleft),
.rotate (shift_rotate),
.arith (shift_arith),
.dout (shift_dout)
);
reg [W_DATA-1:0] op_a_rev;
always @ (*) begin: rev_op_a
integer i;
for (i = 0; i < W_DATA; i = i + 1) begin
op_a_rev[i] = op_a[W_DATA - 1 - i];
end
end
// "leading" means starting at MSB. This is an LSB-first priority encoder, so
// "leading" is reversed and "trailing" is not.
wire [W_DATA-1:0] ctz_search_mask = aluop == ALUOP_CLZ ? op_a_rev : op_a;
wire [W_SHAMT:0] ctz_clz;
hazard3_priority_encode #(
.W_REQ (W_DATA),
.HIGHEST_WINS (0)
) ctz_priority_encode (
.req (ctz_search_mask),
.gnt (ctz_clz[W_SHAMT-1:0])
);
// Special case: all-zeroes returns XLEN
assign ctz_clz[W_SHAMT] = ~|op_a;
reg [W_SHAMT:0] cpop;
always @ (*) begin: cpop_count
integer i;
cpop = {W_SHAMT+1{1'b0}};
for (i = 0; i < W_DATA; i = i + 1) begin
cpop = cpop + {{W_SHAMT{1'b0}}, op_a[i]};
end
end
reg [2*W_DATA-1:0] clmul64;
always @ (*) begin: clmul_mul
integer i;
clmul64 = {2*W_DATA{1'b0}};
for (i = 0; i < W_DATA; i = i + 1) begin
clmul64 = clmul64 ^ (({{W_DATA{1'b0}}, op_a} << i) & {2*W_DATA{op_b[i]}});
end
end
// funct3: 1=clmul, 2=clmulr, 3=clmulh, never 0.
wire [W_DATA-1:0] clmul =
!funct3_32b[1] ? clmul64[31: 0] :
!funct3_32b[0] ? clmul64[62:31] : clmul64[63:32];
reg [W_DATA-1:0] zip;
reg [W_DATA-1:0] unzip;
always @ (*) begin: do_zip_unzip
integer i;
for (i = 0; i < W_DATA; i = i + 1) begin
zip[i] = op_a[{i[0], i[4:1]}]; // Alternate high/low halves
unzip[i] = op_a[{i[3:0], i[4]}]; // All even then all odd
end
end
reg [W_DATA-1:0] xperm8;
always @ (*) begin: do_xperm8
integer i;
for (i = 0; i < W_DATA; i = i + 8) begin
if (|op_b[i + 2 +: 6]) begin
xperm8[i +: 8] = 8'h00;
end else begin
xperm8[i +: 8] = op_a[8 * op_b[i +: 2] +: 8];
end
end
end
reg [W_DATA-1:0] xperm4;
always @ (*) begin: do_xperm4
integer i;
for (i = 0; i < W_DATA; i = i + 4) begin
if (op_b[i + 3]) begin
xperm4[i +: 4] = 4'h0;
end else begin
xperm4[i +: 4] = op_a[4 * op_b[i +: 3] +: 4];
end
end
end
// ----------------------------------------------------------------------------
// Output mux, with simple operations inline
// iCE40: We can implement all bitwise ops with 1 LUT4/bit total, since each
// result bit uses only two operand bits. Much better than feeding each into
// main mux tree. Doesn't matter for big-LUT FPGAs or for implementations with
// bitmanip extensions enabled.
reg [W_DATA-1:0] bitwise;
always @ (*) begin: bitwise_ops
case (aluop[1:0])
ALUOP_AND[1:0]: bitwise = op_a & op_b_inv;
ALUOP_OR [1:0]: bitwise = op_a | op_b_inv;
ALUOP_XOR[1:0]: bitwise = op_a ^ op_b_inv;
ALUOP_RS2[1:0]: bitwise = op_b_inv;
endcase
end
wire [W_DATA-1:0] zbs_mask = {{W_DATA-1{1'b0}}, 1'b1} << op_b[W_SHAMT-1:0];
always @ (*) begin
casez ({|EXTENSION_A, |EXTENSION_ZBA, |EXTENSION_ZBB, |EXTENSION_ZBC,
|EXTENSION_ZBS, |EXTENSION_ZBKB, |EXTENSION_ZBKX, |EXTENSION_XH3BEXTM, aluop})
// Base ISA
{8'bzzzzzzzz, ALUOP_ADD }: result = sum;
{8'bzzzzzzzz, ALUOP_SUB }: result = sum;
{8'bzzzzzzzz, ALUOP_LT }: result = {{W_DATA-1{1'b0}}, lt};
{8'bzzzzzzzz, ALUOP_LTU }: result = {{W_DATA-1{1'b0}}, lt};
{8'bzzzzzzzz, ALUOP_SRL }: result = shift_dout;
{8'bzzzzzzzz, ALUOP_SRA }: result = shift_dout;
{8'bzzzzzzzz, ALUOP_SLL }: result = shift_dout;
// A or Zbb (written this way to avoid case overlap)
{8'b1zzzzzzz, ALUOP_MAX },
{8'b0z1zzzzz, ALUOP_MAX }: result = lt ? op_b : op_a;
{8'b1zzzzzzz, ALUOP_MIN },
{8'b0z1zzzzz, ALUOP_MIN }: result = lt ? op_a : op_b;
{8'b1zzzzzzz, ALUOP_MAXU },
{8'b0z1zzzzz, ALUOP_MAXU }: result = lt ? op_b : op_a;
{8'b1zzzzzzz, ALUOP_MINU },
{8'b0z1zzzzz, ALUOP_MINU }: result = lt ? op_a : op_b;
// Zba
{8'bz1zzzzzz, ALUOP_SHXADD }: result = sum;
// Zbb
{8'bzz1zzzzz, ALUOP_ANDN }: result = bitwise;
{8'bzz1zzzzz, ALUOP_ORN }: result = bitwise;
{8'bzz1zzzzz, ALUOP_XNOR }: result = bitwise;
{8'bzz1zzzzz, ALUOP_CLZ }: result = {{W_DATA-W_SHAMT-1{1'b0}}, ctz_clz};
{8'bzz1zzzzz, ALUOP_CTZ }: result = {{W_DATA-W_SHAMT-1{1'b0}}, ctz_clz};
{8'bzz1zzzzz, ALUOP_CPOP }: result = {{W_DATA-W_SHAMT-1{1'b0}}, cpop};
{8'bzz1zzzzz, ALUOP_SEXT_B }: result = {{W_DATA-8{op_a[7]}}, op_a[7:0]};
{8'bzz1zzzzz, ALUOP_SEXT_H }: result = {{W_DATA-16{op_a[15]}}, op_a[15:0]};
{8'bzz1zzzzz, ALUOP_ZEXT_H }: result = {{W_DATA-16{1'b0}}, op_a[15:0]};
{8'bzz1zzzzz, ALUOP_ORC_B }: result = {{8{|op_a[31:24]}}, {8{|op_a[23:16]}}, {8{|op_a[15:8]}}, {8{|op_a[7:0]}}};
{8'bzz1zzzzz, ALUOP_REV8 }: result = {op_a[7:0], op_a[15:8], op_a[23:16], op_a[31:24]};
{8'bzz1zzzzz, ALUOP_ROL }: result = shift_dout;
{8'bzz1zzzzz, ALUOP_ROR }: result = shift_dout;
// Zbc
{8'bzzz1zzzz, ALUOP_CLMUL }: result = clmul;
// Zbs
{8'bzzzz1zzz, ALUOP_BCLR }: result = op_a & ~zbs_mask;
{8'bzzzz1zzz, ALUOP_BSET }: result = op_a | zbs_mask;
{8'bzzzz1zzz, ALUOP_BINV }: result = op_a ^ zbs_mask;
{8'bzzzz1zzz, ALUOP_BEXT }: result = {{W_DATA-1{1'b0}}, shift_dout[0]};
// Zbkb
{8'bzzzzz1zz, ALUOP_PACK }: result = {op_b[15:0], op_a[15:0]};
{8'bzzzzz1zz, ALUOP_PACKH }: result = {{W_DATA-16{1'b0}}, op_b[7:0], op_a[7:0]};
{8'bzzzzz1zz, ALUOP_BREV8 }: result = {op_a_rev[7:0], op_a_rev[15:8], op_a_rev[23:16], op_a_rev[31:24]};
{8'bzzzzz1zz, ALUOP_UNZIP }: result = unzip;
{8'bzzzzz1zz, ALUOP_ZIP }: result = zip;
// Zbkx
{8'bzzzzzz1z, ALUOP_XPERM }: result = funct3_32b[2] ? xperm8 : xperm4;
// Xh3bextm
{8'bzzzzzzz1, ALUOP_BEXTM }: result = shift_dout & {24'h0, {~(8'hfe << funct7_32b[3:1])}};
default: result = bitwise;
endcase
end
// ----------------------------------------------------------------------------
// Properties for base-ISA instructions
`ifdef HAZARD3_ASSERTIONS
`ifndef RISCV_FORMAL
// Really we're just interested in the shifts and comparisons, as these are
// the nontrivial ones. However, easier to test everything!
wire clk;
always @ (posedge clk) begin
case(aluop)
default: begin end
ALUOP_ADD: assert(result == op_a + op_b);
ALUOP_SUB: assert(result == op_a - op_b);
ALUOP_LT: assert(result == $signed(op_a) < $signed(op_b));
ALUOP_LTU: assert(result == op_a < op_b);
ALUOP_AND: assert(result == (op_a & op_b));
ALUOP_OR: assert(result == (op_a | op_b));
ALUOP_XOR: assert(result == (op_a ^ op_b));
ALUOP_SRL: assert(result == op_a >> op_b[4:0]);
ALUOP_SRA: assert($signed(result) == $signed(op_a) >>> $signed(op_b[4:0]));
ALUOP_SLL: assert(result == op_a << op_b[4:0]);
endcase
end
`endif
`endif
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+52
View File
@@ -0,0 +1,52 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
`default_nettype none
// The branch decision path through the ALU is slow because:
//
// - Sees immediates and PC on its inputs, as well as regs
// - Add/sub rather than just add (with complex decode of the sub condition)
// - 2 extra mux layers in front of adder if Zba extension is enabled
//
// So there is sometimes timing benefit to a dedicated branch comparator.
module hazard3_branchcmp #(
`include "hazard3_config.vh"
,
`include "hazard3_width_const.vh"
) (
input wire [31:0] cir,
input wire [W_DATA-1:0] op_a,
input wire [W_DATA-1:0] op_b,
output wire cmp
);
`include "hazard3_ops.vh"
wire [W_DATA-1:0] diff = op_a - op_b;
// funct3 instruction
// ------------------
// 000 BEQ
// 001 BNE
// 100 BLT
// 101 BGE
// 110 BLTU
// 111 BGEU
wire cmp_is_unsigned = cir[13];
wire lt = op_a[W_DATA-1] == op_b[W_DATA-1] ? diff[W_DATA-1] :
cmp_is_unsigned ? op_b[W_DATA-1] :
op_a[W_DATA-1] ;
assign cmp = cir[14] ? lt : op_a != op_b;
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+162
View File
@@ -0,0 +1,162 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// MUL-only (cfg: MUL_FAST) and MUL/MULH/MULHU/MULHSU (cfg: MUL_FAST &&
// MULH_FAST) are handled by different circuits. In either case it's a simple
// behavioural multiply, and we rely on inference to get good performance on
// FPGA.
`default_nettype none
module hazard3_mul_fast #(
`include "hazard3_config.vh"
,
`include "hazard3_width_const.vh"
) (
input wire clk,
input wire rst_n,
input wire [W_MULOP-1:0] op,
input wire op_vld,
input wire [W_DATA-1:0] op_a,
input wire [W_DATA-1:0] op_b,
output wire [W_DATA-1:0] result,
output reg result_vld
);
`include "hazard3_ops.vh"
localparam XLEN = W_DATA;
//synthesis translate_off
generate if (MULH_FAST && !MUL_FAST) begin: err_require_mul_fast_for_mulh
initial $fatal("%m: MULH_FAST requires that MUL_FAST is also set.");
end endgenerate
generate if (MUL_FASTER && !MUL_FAST) begin: err_require_mul_fast_for_faster
initial $fatal("%m: MUL_FASTER requires that MUL_FAST is also set.");
end endgenerate
//synthesis translate_on
// Latency of 1:
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
result_vld <= 1'b0;
end else begin
result_vld <= op_vld;
end
end
// ----------------------------------------------------------------------------
// Fast MUL only
generate
if (!MULH_FAST) begin: mul_only
// This pipestage is folded into the front of the DSP tiles on UP5k. Note the
// intention is to register the bypassed core regs at the end of X (since
// bypass is quite slow), then perform multiply combinatorially in stage M,
// and mux into MW result register.
reg [XLEN-1:0] op_a_r;
reg [XLEN-1:0] op_b_r;
if (MUL_FASTER) begin: op_passthrough
always @ (*) begin
op_a_r = op_a;
op_b_r = op_b;
end
end else begin: op_register
always @ (posedge clk) begin
if (op_vld) begin
op_a_r <= op_a;
op_b_r <= op_b;
end
end
end
// This should be inferred as 3 DSP tiles on UP5k:
//
// 1. Register then multiply a[15: 0] and b[15: 0]
// 2. Register then multiply a[31:16] and b[15: 0], then directly add output of 1
// 3. Register then multiply a[15: 0] and b[31:16], then directly add output of 2
//
// So there is quite a long path (1x 16-bit multiply, then 2x 16-bit add). On
// other platforms you may just end up with a pile of gates.
`ifndef RISCV_FORMAL_ALTOPS
assign result = op_a_r * op_b_r;
`else
// riscv-formal can use a simpler function, since it's just confirming the
// result is correctly hooked up.
assign result = result_vld ? (op_a_r + op_b_r) ^ 32'h5876063e : 32'hdeadbeef;
`endif
// ----------------------------------------------------------------------------
// Fast MUL/MULH/MULHU/MULHSU
end else begin: mul_and_mulh
reg [XLEN-1:0] op_a_r;
reg [XLEN-1:0] op_b_r;
reg [W_MULOP-1:0] op_r;
if (MUL_FASTER) begin: op_passthrough
always @ (*) begin
op_a_r = op_a;
op_b_r = op_b;
op_r = op;
end
end else begin: op_register
always @ (posedge clk) begin
if (op_vld) begin
op_a_r <= op_a;
op_b_r <= op_b;
op_r <= op;
end
end
end
wire op_a_signed = op_r == M_OP_MULH || op_r == M_OP_MULHSU;
wire op_b_signed = op_r == M_OP_MULH;
wire [2*XLEN-1:0] op_a_sext = {
{XLEN{op_a_r[XLEN - 1] && op_a_signed}},
op_a_r
};
wire [2*XLEN-1:0] op_b_sext = {
{XLEN{op_b_r[XLEN - 1] && op_b_signed}},
op_b_r
};
wire [2*XLEN-1:0] result_full = op_a_sext * op_b_sext;
`ifndef RISCV_FORMAL_ALTOPS
assign result = op_r == M_OP_MUL ? result_full[0 +: XLEN] : result_full[XLEN +: XLEN];
`else
assign result =
op_r == M_OP_MULH ? (op_a_r + op_b_r) ^ 32'hf6583fb7 :
op_r == M_OP_MULHSU ? (op_a_r - op_b_r) ^ 32'hecfbe137 :
op_r == M_OP_MULHU ? (op_a_r + op_b_r) ^ 32'h949ce5e8 :
op_r == M_OP_MUL ? (op_a_r + op_b_r) ^ 32'h5876063e : 32'hdeadbeef;
`endif
end
endgenerate
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+294
View File
@@ -0,0 +1,294 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Combined multiply/divide/modulo circuit. All operations performed at 1 bit
// per clock; aiming for minimal resource usage on iCE40 FPGA. Optionally the
// circuit can be unrolled for slightly higher performance.
//
// When op_kill is high, the current calculation halts immediately. op_vld can
// be asserted on the same cycle, and the new calculation begins without
// delay, regardless of op_rdy. This may be used by the processor on e.g.
// mispredict or trap.
//
// The actual multiply/divide hardware is unsigned. We handle signedness at
// input/output.
`default_nettype none
module hazard3_muldiv_seq #(
`include "hazard3_config.vh"
,
`include "hazard3_width_const.vh"
) (
input wire clk,
input wire rst_n,
input wire [W_MULOP-1:0] op,
input wire op_vld,
output wire op_rdy,
input wire op_kill,
input wire [W_DATA-1:0] op_a,
input wire [W_DATA-1:0] op_b,
output wire [W_DATA-1:0] result_h, // mulh* or rem*
output wire [W_DATA-1:0] result_l, // mul or div*
output wire result_vld
);
`include "hazard3_ops.vh"
//synthesis translate_off
generate if (|(MULDIV_UNROLL & (MULDIV_UNROLL - 1)) || ~|MULDIV_UNROLL) begin: err_pow2
initial $fatal("%m: MULDIV_UNROLL must be a positive power of 2");
end endgenerate
//synthesis translate_on
localparam XLEN = W_DATA;
parameter W_CTR = $clog2(XLEN + 1);
// ----------------------------------------------------------------------------
// Operation decode, operand sign adjustment
// On the first cycle, op_a and op_b go straight through to the accumulator
// and the divisor/multiplicand register. They are then adjusted in-place
// on the next cycle. This allows the same circuits to be reused for sign
// adjustment before output (and helps input timing).
reg [W_MULOP-1:0] op_r;
reg [2*XLEN-1:0] accum;
reg [XLEN-1:0] op_b_r;
reg op_a_neg_r;
reg op_b_neg_r;
wire op_a_signed =
op_r == M_OP_MULH ||
op_r == M_OP_MULHSU ||
op_r == M_OP_DIV ||
op_r == M_OP_REM;
wire op_b_signed =
op_r == M_OP_MULH ||
op_r == M_OP_DIV ||
op_r == M_OP_REM;
wire op_a_neg = op_a_signed && accum[XLEN-1];
wire op_b_neg = op_b_signed && op_b_r[XLEN-1];
// Non-divide parts of the circuit should be constant-folded if all the MUL
// operations are handled by the fast multiplier
wire is_div = op_r[2] || (MUL_FAST && MULH_FAST);
// Controls for modifying sign of all/part of accumulator
wire accum_neg_l;
wire accum_inv_h;
wire accum_incr_h;
// ----------------------------------------------------------------------------
// Arithmetic circuit
// Combinatorials:
reg [2*XLEN-1:0] accum_next;
reg [2*XLEN-1:0] addend;
reg [2*XLEN-1:0] shift_tmp;
reg [2*XLEN-1:0] addsub_tmp;
reg neg_l_borrow;
always @ (*) begin: alu
integer i;
// Multiply/divide iteration layers
accum_next = accum;
addend = {2*XLEN{1'b0}};
addsub_tmp = {2*XLEN{1'b0}};
neg_l_borrow = 1'b0;
for (i = 0; i < MULDIV_UNROLL; i = i + 1) begin
addend = {is_div && |op_b_r, op_b_r, {XLEN-1{1'b0}}};
shift_tmp = is_div ? accum_next : accum_next >> 1;
addsub_tmp = shift_tmp + addend;
accum_next = (is_div ? !addsub_tmp[2 * XLEN - 1] : accum_next[0]) ?
addsub_tmp : shift_tmp;
if (is_div)
accum_next = {accum_next[2*XLEN-2:0], !addsub_tmp[2 * XLEN - 1]};
end
// Alternative path for negation of all/part of accumulator
if (accum_neg_l)
{neg_l_borrow, accum_next[XLEN-1:0]} = {~accum[XLEN-1:0]} + 1'b1;
if (accum_incr_h || accum_inv_h)
accum_next[XLEN +: XLEN] = (accum[XLEN +: XLEN] ^ {XLEN{accum_inv_h}})
+ {{XLEN-1{1'b0}}, accum_incr_h};
end
// ----------------------------------------------------------------------------
// Main state machine
reg sign_preadj_done;
reg [W_CTR-1:0] ctr;
reg sign_postadj_done;
reg sign_postadj_carry;
localparam CTR_TOP = XLEN[W_CTR-1:0];
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
ctr <= {W_CTR{1'b0}};
sign_preadj_done <= 1'b1;
sign_postadj_done <= 1'b1;
sign_postadj_carry <= 1'b0;
op_r <= {W_MULOP{1'b0}};
op_a_neg_r <= 1'b0;
op_b_neg_r <= 1'b0;
op_b_r <= {XLEN{1'b0}};
accum <= {XLEN*2{1'b0}};
end else if (op_kill || (op_vld && op_rdy)) begin
// Initialise circuit with operands + state
ctr <= op_vld ? CTR_TOP : {W_CTR{1'b0}};
sign_preadj_done <= !op_vld;
sign_postadj_done <= !op_vld;
sign_postadj_carry <= 1'b0;
op_r <= op;
op_b_r <= op_b;
accum <= {{XLEN{1'b0}}, op_a};
end else if (!sign_preadj_done) begin
// Pre-adjust sign if necessary, else perform first iteration immediately
op_a_neg_r <= op_a_neg;
op_b_neg_r <= op_b_neg;
sign_preadj_done <= 1'b1;
if (accum_neg_l || (op_b_neg ^ is_div)) begin
if (accum_neg_l)
accum[0 +: XLEN] <= accum_next[0 +: XLEN];
if (op_b_neg ^ is_div)
op_b_r <= -op_b_r;
end else begin
ctr <= ctr - MULDIV_UNROLL[W_CTR-1:0];
accum <= accum_next;
end
end else if (|ctr) begin
ctr <= ctr - MULDIV_UNROLL[W_CTR-1:0];
accum <= accum_next;
end else if (!sign_postadj_done || sign_postadj_carry) begin
sign_postadj_done <= 1'b1;
if (accum_inv_h || accum_incr_h)
accum[XLEN +: XLEN] <= accum_next[XLEN +: XLEN];
if (accum_neg_l) begin
accum[0 +: XLEN] <= accum_next[0 +: XLEN];
if (!is_div) begin
sign_postadj_carry <= neg_l_borrow;
sign_postadj_done <= !neg_l_borrow;
end
end
end
end
// ----------------------------------------------------------------------------
// Sign adjustment control
// Pre-adjustment: for any a, b we want |a|, |b|. Note that the magnitude of any
// 32-bit signed integer is representable by a 32-bit unsigned integer.
// Post-adjustment for division:
// We seek q, r to satisfy a = b * q + r, where a and b are given,
// and |r| < |b|. One way to do this is if
// sgn(r) = sgn(a)
// sgn(q) = sgn(a) ^ sgn(b)
// This has additional nice properties like
// -(a / b) = (-a) / b = a / (-b)
// Post-adjustment for multiplication:
// We have calculated the 2*XLEN result of |a| * |b|.
// Negate the entire accumulator if sgn(a) ^ sgn(b).
// This is done in two steps (to share div/mod circuit, and avoid 64-bit carry):
// - Negate lower half of accumulator, and invert upper half
// - Increment upper half if lower half carried
wire do_postadj = ~|{ctr, sign_postadj_done};
wire op_signs_differ = op_a_neg_r ^ op_b_neg_r;
assign accum_neg_l =
!sign_preadj_done && op_a_neg ||
do_postadj && !sign_postadj_carry && op_signs_differ && !(is_div && ~|op_b_r);
assign {accum_incr_h, accum_inv_h} =
do_postadj && is_div && op_a_neg_r ? 2'b11 :
do_postadj && !is_div && op_signs_differ && !sign_postadj_carry ? 2'b01 :
do_postadj && !is_div && op_signs_differ && sign_postadj_carry ? 2'b10 :
2'b00 ;
// ----------------------------------------------------------------------------
// Outputs
assign op_rdy = ~|{ctr, accum_neg_l, accum_incr_h, accum_inv_h};
assign result_vld = op_rdy;
`ifndef RISCV_FORMAL_ALTOPS
assign {result_h, result_l} = accum;
`else
// Provide arithmetically simpler alternative operations, to speed up formal checks
always assert(XLEN == 32);
reg [XLEN-1:0] fml_a_saved;
reg [XLEN-1:0] fml_b_saved;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
fml_a_saved <= {XLEN{1'b0}};
fml_b_saved <= {XLEN{1'b0}};
end else if (op_vld && op_rdy) begin
fml_a_saved <= op_a;
fml_b_saved <= op_b;
end
end
assign result_h =
op_r == M_OP_MULH ? (fml_a_saved + fml_b_saved) ^ 32'hf6583fb7 :
op_r == M_OP_MULHSU ? (fml_a_saved - fml_b_saved) ^ 32'hecfbe137 :
op_r == M_OP_MULHU ? (fml_a_saved + fml_b_saved) ^ 32'h949ce5e8 :
op_r == M_OP_REM ? (fml_a_saved - fml_b_saved) ^ 32'h8da68fa5 :
op_r == M_OP_REMU ? (fml_a_saved - fml_b_saved) ^ 32'h3138d0e1 : 32'hdeadbeef;
assign result_l =
op_r == M_OP_MUL ? (fml_a_saved + fml_b_saved) ^ 32'h5876063e :
op_r == M_OP_DIV ? (fml_a_saved - fml_b_saved) ^ 32'h7f8529ec :
op_r == M_OP_DIVU ? (fml_a_saved - fml_b_saved) ^ 32'h10e8fd70 : 32'hdeadbeef;
`endif
// ----------------------------------------------------------------------------
// Interface properties
`ifdef HAZARD3_ASSERTIONS
always @ (posedge clk) if (rst_n && $past(rst_n)) begin: properties
integer i;
reg alive;
if ($past(op_rdy && !op_vld))
assert(op_rdy);
if (result_vld && $past(result_vld) && !$past(op_kill))
assert($stable({result_h, result_l}));
// Kill will halt an in-progress operation, but a new operation may be
// asserted simultaneously with kill.
if ($past(op_kill))
assert(op_rdy == !$past(op_vld));
// We should be periodically ready (liveness property), unless new operations
// are forced in immediately, simultaneous with a kill, in which case there
// is no intermediate ready state.
alive = op_rdy || (op_kill && op_vld);
for (i = 1; i <= XLEN / MULDIV_UNROLL + 3; i = i + 1)
alive = alive || $past(op_rdy || (op_kill && op_vld), i);
assert(alive);
end
`endif
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+31
View File
@@ -0,0 +1,31 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// req: one-hot bitmap
// idx: index of the sole set bit in req
`default_nettype none
module hazard3_onehot_encode #(
parameter W_REQ = 16,
parameter W_GNT = $clog2(W_REQ) // do not modify
) (
input wire [W_REQ-1:0] req,
output reg [W_GNT-1:0] gnt
);
always @ (*) begin: encode
reg [W_GNT:0] i;
gnt = {W_GNT{1'b0}};
for (i = 0; i < W_REQ; i = i + 1) begin
gnt = gnt | ({W_GNT{req[i[W_GNT-1:0]]}} & i[W_GNT-1:0]);
end
end
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+33
View File
@@ -0,0 +1,33 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// req: bitmap
// idx: bitmap with all bits clear except the least- (HIGHEST_WINS=0) or
// most- (HIGHEST_WINS=1) significant set bit in req.
`default_nettype none
module hazard3_onehot_priority #(
parameter W_REQ = 16,
parameter HIGHEST_WINS = 0
) (
input wire [W_REQ-1:0] req,
output reg [W_REQ-1:0] gnt
);
always @ (*) begin: select
integer i;
for (i = 0; i < W_REQ; i = i + 1) begin
gnt[i] = req[i] && ~|(req & (
HIGHEST_WINS ? ~({W_REQ{1'b1}} >> (W_REQ - 1 - i)) : ~({W_REQ{1'b1}} << i)
));
end
end
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
@@ -0,0 +1,77 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// req: bitmap of requests
// priority: packed array of dynamic priority level of each request
// gnt: one-hot bitmap with the highest-priority request.
`default_nettype none
module hazard3_onehot_priority_dynamic #(
parameter W_REQ = 8,
parameter N_PRIORITIES = 2,
parameter PRIORITY_HIGHEST_WINS = 1, // If 1, numerically highest level has greatest priority.
// Otherwise, numerically lowest wins.
parameter TIEBREAK_HIGHEST_WINS = 0, // If 1, highest-numbered request at the highest priority
// level wins the tiebreak. Otherwise, lowest-numbered.
// Do not modify:
parameter W_PRIORITY = $clog2(N_PRIORITIES)
) (
input wire [W_REQ*W_PRIORITY-1:0] pri,
input wire [W_REQ-1:0] req,
output wire [W_REQ-1:0] gnt
);
// 1. Stratify requests according to their level
reg [W_REQ-1:0] req_stratified [0:N_PRIORITIES-1];
reg [N_PRIORITIES-1:0] level_has_req;
always @ (*) begin: stratify
reg signed [31:0] i, j;
for (i = 0; i < N_PRIORITIES; i = i + 1) begin
for (j = 0; j < W_REQ; j = j + 1) begin
req_stratified[i][j] = req[j] &&
pri[W_PRIORITY * j +: W_PRIORITY] == i[W_PRIORITY-1:0];
end
level_has_req[i] = |req_stratified[i];
end
end
// 2. Select the highest level with active requests
wire [N_PRIORITIES-1:0] active_layer_sel;
hazard3_onehot_priority #(
.W_REQ (N_PRIORITIES),
.HIGHEST_WINS (PRIORITY_HIGHEST_WINS)
) prisel_layer (
.req (level_has_req),
.gnt (active_layer_sel)
);
// 3. Mask only those requests at this level
reg [W_REQ-1:0] reqs_from_highest_layer;
always @ (*) begin: mux_reqs_by_layer
integer i;
reqs_from_highest_layer = {W_REQ{1'b0}};
for (i = 0; i < N_PRIORITIES; i = i + 1)
reqs_from_highest_layer = reqs_from_highest_layer |
(req_stratified[i] & {W_REQ{active_layer_sel[i]}});
end
// 4. Do a standard priority select on those requests as a tie break
hazard3_onehot_priority #(
.W_REQ (W_REQ),
.HIGHEST_WINS (TIEBREAK_HIGHEST_WINS)
) prisel_tiebreak (
.req (reqs_from_highest_layer),
.gnt (gnt)
);
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+41
View File
@@ -0,0 +1,41 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// req: bitmap
// gnt: index of least set bit (HIGHEST_WINS=0) or most set bit (HIGHEST_WINS=1)
`default_nettype none
module hazard3_priority_encode #(
parameter W_REQ = 16,
parameter HIGHEST_WINS = 0,
parameter W_GNT = $clog2(W_REQ) // do not modify
) (
input wire [W_REQ-1:0] req,
output wire [W_GNT-1:0] gnt
);
wire [W_REQ-1:0] gnt_onehot;
hazard3_onehot_priority #(
.W_REQ (W_REQ),
.HIGHEST_WINS (HIGHEST_WINS)
) priority_u (
.req (req),
.gnt (gnt_onehot)
);
hazard3_onehot_encode #(
.W_REQ (W_REQ)
) encode_u (
.req (gnt_onehot),
.gnt (gnt)
);
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+66
View File
@@ -0,0 +1,66 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Implement the three shifts (left logical, right logical, right arithmetic)
// using a single log-type barrel shifter. Around 240 LUTs for 32 bits.
// (7 layers of 32 2-input muxes, some extra LUTs and LUT inputs used for arith)
`default_nettype none
module hazard3_shift_barrel #(
`include "hazard3_config.vh"
,
`include "hazard3_width_const.vh"
) (
input wire [W_DATA-1:0] din,
input wire [W_SHAMT-1:0] shamt,
input wire right_nleft,
input wire rotate,
input wire arith,
output reg [W_DATA-1:0] dout
);
reg [W_DATA-1:0] din_rev;
reg [W_DATA-1:0] shift_accum;
reg sext; // haha
always @ (*) begin: shift
integer i;
for (i = 0; i < W_DATA; i = i + 1)
din_rev[i] = right_nleft ? din[W_DATA - 1 - i] : din[i];
sext = arith && din_rev[0];
shift_accum = din_rev;
for (i = 0; i < W_SHAMT; i = i + 1) begin
if (shamt[i]) begin
shift_accum = (shift_accum << (1 << i)) |
({W_DATA{sext}} & ~({W_DATA{1'b1}} << (1 << i))) |
({W_DATA{rotate && |EXTENSION_ZBB}} & (shift_accum >> (W_DATA - (1 << i))));
end
end
for (i = 0; i < W_DATA; i = i + 1)
dout[i] = right_nleft ? shift_accum[W_DATA - 1 - i] : shift_accum[i];
end
`ifdef HAZARD3_ASSERTIONS
always @ (*) begin
if (right_nleft && arith && !rotate) begin: asr
assert($signed(dout) == $signed(din) >>> $signed(shamt));
end else if (right_nleft && !arith && !rotate) begin
assert(dout == din >> shamt);
end else if (!right_nleft && !arith && !rotate) begin
assert(dout == din << shamt);
end
end
`endif
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+65
View File
@@ -0,0 +1,65 @@
#!/usr/bin/env python3
# Quick reference model for sequential unsigned multiply/divide/modulo
def div_step(w, accum, divisor):
sub_tmp = accum - (divisor << (w - 1))
underflow = sub_tmp < 0
if not underflow:
accum = sub_tmp
accum = (accum << 1) | (not underflow)
return accum
def divmod(w, dividend, divisor, debug=True):
accum = dividend
for i in range(w):
accum_prev = accum
accum = div_step(w, accum, divisor)
if debug:
print("Step {:02d}: accum {:0{}x} -> {:0{}x}".format(
i, accum_prev, int(w / 2), accum, int(w / 2)))
return (accum >> w, accum & ((1 << w) - 1))
def mul_step(w, accum, multiplicand):
add_en = accum & 1
accum = accum >> 1
if add_en:
accum += (multiplicand << (w - 1))
return accum
def mul(w, multiplicand, multiplier, debug=True):
accum = multiplier
for i in range(w):
accum_prev = accum
accum = mul_step(w, accum, multiplicand)
if debug:
print("Step {:02d}: accum {:0{}x} -> {:0{}x}".format(
i, accum_prev, int(w / 2), accum, int(w / 2)))
return (accum >> w, accum & ((1 << w) - 1))
def divtest(w=4):
for i in range(2 ** w):
for j in range(1, 2 ** w):
gatemod, gatediv = divmod(w, i, j, debug=False)
goldmod, golddiv = (i % j, i // j)
print("{:02d} % {:02d} = {:02d} (gold {:02d}); ./. = {:02d} (gold {:02d})"
.format(i, j, gatemod, goldmod, gatediv, golddiv))
assert(gatemod == goldmod)
assert(gatediv == golddiv)
def multest(w=4):
for i in range(2 ** w):
for j in range(2 ** w):
gateh, gatel = mul(w, i, j, debug=False)
gold = i * j
goldl, goldh = (gold & ((1 << w) - 1), gold >> w)
print("{:02d} * {:02d} = ({:02d} (gold {:02d}), {:02d} (gold {:02d})"
.format(i, j, gateh, goldh, gatel, goldl))
assert(gatel == goldl)
assert(gateh == goldh)
if __name__ == "__main__":
print("Test division:")
divtest()
print("Test multiplication:")
multest()
+200
View File
@@ -0,0 +1,200 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// APB-to-APB asynchronous bridge for connecting DTM to DM, in case DTM is in
// a different clock domain (e.g. running directly from crystal to get a
// fixed baud reference)
//
// Note this module depends on the hazard3_sync_1bit module (a flop-chain
// synchroniser) which should be reimplemented for your FPGA/process.
`ifndef HAZARD3_REG_KEEP_ATTRIBUTE
`define HAZARD3_REG_KEEP_ATTRIBUTE (* keep = 1'b1 *)
`endif
`default_nettype none
module hazard3_apb_async_bridge #(
parameter W_ADDR = 8,
parameter W_DATA = 32,
parameter N_SYNC_STAGES = 2
) (
// Resets assumed to be synchronised externally
input wire clk_src,
input wire rst_n_src,
input wire clk_dst,
input wire rst_n_dst,
// APB port from Transport Module
input wire src_psel,
input wire src_penable,
input wire src_pwrite,
input wire [W_ADDR-1:0] src_paddr,
input wire [W_DATA-1:0] src_pwdata,
output wire [W_DATA-1:0] src_prdata,
output wire src_pready,
output wire src_pslverr,
// APB port to Debug Module
output wire dst_psel,
output wire dst_penable,
output wire dst_pwrite,
output wire [W_ADDR-1:0] dst_paddr,
output wire [W_DATA-1:0] dst_pwdata,
input wire [W_DATA-1:0] dst_prdata,
input wire dst_pready,
input wire dst_pslverr
);
// ----------------------------------------------------------------------------
// Clock-crossing registers
// We're using a modified req/ack handshake:
//
// - Initially both req and ack are low
// - src asserts req high
// - dst responds with ack high and begins transfer
// - src deasserts req once it sees ack high
// - dst deasserts ack once:
// - transfer is complete *and*
// - dst sees req deasserted
// - Once src sees ack low, a new transfer can begin.
//
// A NRZI toggle handshake might be more appropriate, but can cause spurious
// bus accesses when only one side of the link is reset.
`HAZARD3_REG_KEEP_ATTRIBUTE reg src_req;
wire dst_req;
`HAZARD3_REG_KEEP_ATTRIBUTE reg dst_ack;
wire src_ack;
// Note the launch registers are not resettable. We maintain setup/hold on
// launch-to-capture paths thanks to the req/ack handshake. A stray reset
// could violate this.
//
// The req/ack logic itself can be reset safely because the receiving domain
// is protected from metastability by a 2FF synchroniser.
`HAZARD3_REG_KEEP_ATTRIBUTE reg [W_ADDR + W_DATA + 1 -1:0] src_paddr_pwdata_pwrite; // launch
`HAZARD3_REG_KEEP_ATTRIBUTE reg [W_ADDR + W_DATA + 1 -1:0] dst_paddr_pwdata_pwrite; // capture
`HAZARD3_REG_KEEP_ATTRIBUTE reg [W_DATA + 1 -1:0] dst_prdata_pslverr; // launch
`HAZARD3_REG_KEEP_ATTRIBUTE reg [W_DATA + 1 -1:0] src_prdata_pslverr; // capture
hazard3_sync_1bit #(
.N_STAGES (N_SYNC_STAGES)
) sync_req (
.clk (clk_dst),
.rst_n (rst_n_dst),
.i (src_req),
.o (dst_req)
);
hazard3_sync_1bit #(
.N_STAGES (N_SYNC_STAGES)
) sync_ack (
.clk (clk_src),
.rst_n (rst_n_src),
.i (dst_ack),
.o (src_ack)
);
// ----------------------------------------------------------------------------
// src state machine
reg src_waiting_for_downstream;
reg src_pready_r;
always @ (posedge clk_src or negedge rst_n_src) begin
if (!rst_n_src) begin
src_req <= 1'b0;
src_waiting_for_downstream <= 1'b0;
src_prdata_pslverr <= {W_DATA + 1{1'b0}};
src_pready_r <= 1'b1;
end else if (src_waiting_for_downstream) begin
if (src_req && src_ack) begin
// Request was acknowledged, so deassert.
src_req <= 1'b0;
end else if (!(src_req || src_ack)) begin
// Downstream transfer has finished, data is valid.
src_pready_r <= 1'b1;
src_waiting_for_downstream <= 1'b0;
// Note this assignment is cross-domain (but data has been stable
// for duration of ack synchronisation delay):
src_prdata_pslverr <= dst_prdata_pslverr;
end
end else begin
// paddr, pwdata and pwrite are all valid during the setup phase, and
// APB defines the setup phase to always last one cycle and proceed
// to access phase. So, we can ignore penable, and pready is ignored.
if (src_psel) begin
src_pready_r <= 1'b0;
src_req <= 1'b1;
src_waiting_for_downstream <= 1'b1;
end
end
end
// Bus request launch register is not resettable
always @ (posedge clk_src) begin
if (src_psel && !src_waiting_for_downstream)
src_paddr_pwdata_pwrite <= {src_paddr, src_pwdata, src_pwrite};
end
assign {src_prdata, src_pslverr} = src_prdata_pslverr;
assign src_pready = src_pready_r;
// ----------------------------------------------------------------------------
// dst state machine
wire dst_bus_finish = dst_penable && dst_pready;
reg dst_psel_r;
reg dst_penable_r;
always @ (posedge clk_dst or negedge rst_n_dst) begin
if (!rst_n_dst) begin
dst_ack <= 1'b0;
end else if (dst_req) begin
dst_ack <= 1'b1;
end else if (!dst_req && dst_ack && !dst_psel_r) begin
dst_ack <= 1'b0;
end
end
always @ (posedge clk_dst or negedge rst_n_dst) begin
if (!rst_n_dst) begin
dst_psel_r <= 1'b0;
dst_penable_r <= 1'b0;
dst_paddr_pwdata_pwrite <= {W_ADDR + W_DATA + 1{1'b0}};
end else if (dst_req && !dst_ack) begin
dst_psel_r <= 1'b1;
// Note this assignment is cross-domain. The src register has been
// stable for the duration of the req sync delay.
dst_paddr_pwdata_pwrite <= src_paddr_pwdata_pwrite;
end else if (dst_psel_r && !dst_penable_r) begin
dst_penable_r <= 1'b1;
end else if (dst_bus_finish) begin
dst_psel_r <= 1'b0;
dst_penable_r <= 1'b0;
end
end
// Bus response launch register is not resettable
always @ (posedge clk_dst) begin
if (dst_bus_finish)
dst_prdata_pslverr <= {dst_prdata, dst_pslverr};
end
assign dst_psel = dst_psel_r;
assign dst_penable = dst_penable_r;
assign {dst_paddr, dst_pwdata, dst_pwrite} = dst_paddr_pwdata_pwrite;
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+41
View File
@@ -0,0 +1,41 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// The output is asserted asynchronously when the input is asserted,
// but deasserted synchronously when clocked with the input deasserted.
// Input and output are both active-low.
//
// This is a baseline implementation -- you should replace it with cells
// specific to your FPGA/process
`ifndef HAZARD3_REG_KEEP_ATTRIBUTE
`define HAZARD3_REG_KEEP_ATTRIBUTE (* keep = 1'b1 *)
`endif
`default_nettype none
module hazard3_reset_sync #(
parameter N_STAGES = 2 // Should be >= 2
) (
input wire clk,
input wire rst_n_in,
output wire rst_n_out
);
`HAZARD3_REG_KEEP_ATTRIBUTE reg [N_STAGES-1:0] delay;
always @ (posedge clk or negedge rst_n_in)
if (!rst_n_in)
delay <= {N_STAGES{1'b0}};
else
delay <= {delay[N_STAGES-2:0], 1'b1};
assign rst_n_out = delay[N_STAGES-1];
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+39
View File
@@ -0,0 +1,39 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// A 2FF synchronizer to mitigate metastabilities. This is a baseline
// implementation -- you should replace it with cells specific to your
// FPGA/process
`ifndef HAZARD3_REG_KEEP_ATTRIBUTE
`define HAZARD3_REG_KEEP_ATTRIBUTE (* keep = 1'b1 *) (* async_reg *)
`endif
`default_nettype none
module hazard3_sync_1bit #(
parameter N_STAGES = 2 // Should be >=2
) (
input wire clk,
input wire rst_n,
input wire i,
output wire o
);
`HAZARD3_REG_KEEP_ATTRIBUTE reg [N_STAGES-1:0] sync_flops;
always @ (posedge clk or negedge rst_n)
if (!rst_n)
sync_flops <= {N_STAGES{1'b0}};
else
sync_flops <= {sync_flops[N_STAGES-2:0], i};
assign o = sync_flops[N_STAGES-1];
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+1
View File
@@ -0,0 +1 @@
file hazard3_dm.v
+904
View File
@@ -0,0 +1,904 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// RISC-V Debug Module for Hazard3. Supports up to 32 cores (1 hart per core).
`default_nettype none
module hazard3_dm #(
// Where there are multiple harts per DM, the least-indexed hart is the
// least-significant on each concatenated hart access bus.
parameter N_HARTS = 1,
// Where there are multiple DMs, the address of each DM should be a
// multiple of 'h200, so that bits[8:2] decode correctly.
parameter NEXT_DM_ADDR = 32'h0000_0000,
// Implement support for system bus access:
parameter HAVE_SBA = 0,
// Do not modify:
parameter XLEN = 32, // Do not modify
parameter W_HARTSEL = N_HARTS > 1 ? $clog2(N_HARTS) : 1 // Do not modify
) (
// DM is assumed to be in same clock domain as core; clock crossing
// (if any) is inside DTM, or between DTM and DM.
input wire clk,
input wire rst_n,
// APB access from Debug Transport Module
input wire dmi_psel,
input wire dmi_penable,
input wire dmi_pwrite,
input wire [8:0] dmi_paddr,
input wire [31:0] dmi_pwdata,
output reg [31:0] dmi_prdata,
output wire dmi_pready,
output wire dmi_pslverr,
// Reset request/acknowledge. "req" is a pulse >= 1 cycle wide. "done" is
// level-sensitive, goes high once component is out of reset.
//
// The "sys" reset (ndmreset) is conventionally everything apart from DM +
// DTM, but, as per section 3.2 in 0.13.2 debug spec: "Exactly what is
// affected by this reset is implementation dependent, as long as it is
// possible to debug programs from the first instruction executed." So
// this could simply be an all-hart reset.
output wire sys_reset_req,
input wire sys_reset_done,
output wire [N_HARTS-1:0] hart_reset_req,
input wire [N_HARTS-1:0] hart_reset_done,
// Hart run/halt control
output wire [N_HARTS-1:0] hart_req_halt,
output wire [N_HARTS-1:0] hart_req_halt_on_reset,
output wire [N_HARTS-1:0] hart_req_resume,
input wire [N_HARTS-1:0] hart_halted,
input wire [N_HARTS-1:0] hart_running,
// Hart access to data0 CSR (assumed to be core-internal but per-hart)
output wire [N_HARTS*XLEN-1:0] hart_data0_rdata,
input wire [N_HARTS*XLEN-1:0] hart_data0_wdata,
input wire [N_HARTS-1:0] hart_data0_wen,
// Hart instruction injection
output wire [N_HARTS*32-1:0] hart_instr_data,
output reg [N_HARTS-1:0] hart_instr_data_vld,
input wire [N_HARTS-1:0] hart_instr_data_rdy,
input wire [N_HARTS-1:0] hart_instr_caught_exception,
input wire [N_HARTS-1:0] hart_instr_caught_ebreak,
// System bus access (optional) -- can be hooked up to the standalone AHB
// shim (hazard3_sbus_to_ahb.v) or the SBA input port on the processor
// wrapper, which muxes SBA into the processor's load/store bus access
// port. SBA does not increase debugger bus throughput, but supports
// minimally intrusive debug bus access for e.g. Segger RTT.
output wire [31:0] sbus_addr,
output wire sbus_write,
output wire [1:0] sbus_size,
output wire sbus_vld,
input wire sbus_rdy,
input wire sbus_err,
output wire [31:0] sbus_wdata,
input wire [31:0] sbus_rdata
);
wire dmi_write = dmi_psel && dmi_penable && dmi_pready && dmi_pwrite;
wire dmi_read = dmi_psel && dmi_penable && dmi_pready && !dmi_pwrite;
assign dmi_pready = 1'b1;
assign dmi_pslverr = 1'b0;
// Program buffer is fixed at 2 words plus impebreak. The main thing we care
// about is support for efficient memory block transfers using abstractauto;
// in this case 2 words + impebreak is sufficient for RV32I, and 1 word +
// impebreak is sufficient for RV32IC.
localparam PROGBUF_SIZE = 2;
// ----------------------------------------------------------------------------
// Address constants
localparam ADDR_DATA0 = 7'h04;
// Other data registers not present.
localparam ADDR_DMCONTROL = 7'h10;
localparam ADDR_DMSTATUS = 7'h11;
localparam ADDR_HARTINFO = 7'h12;
localparam ADDR_HALTSUM1 = 7'h13;
localparam ADDR_HALTSUM0 = 7'h40;
// No HALTSUM2+ registers (we don't support >32 harts anyway)
localparam ADDR_HAWINDOWSEL = 7'h14;
localparam ADDR_HAWINDOW = 7'h15;
localparam ADDR_ABSTRACTCS = 7'h16;
localparam ADDR_COMMAND = 7'h17;
localparam ADDR_ABSTRACTAUTO = 7'h18;
localparam ADDR_CONFSTRPTR0 = 7'h19;
localparam ADDR_CONFSTRPTR1 = 7'h1a;
localparam ADDR_CONFSTRPTR2 = 7'h1b;
localparam ADDR_CONFSTRPTR3 = 7'h1c;
localparam ADDR_NEXTDM = 7'h1d;
localparam ADDR_PROGBUF0 = 7'h20;
localparam ADDR_PROGBUF1 = 7'h21;
// No authentication
localparam ADDR_SBCS = 7'h38;
localparam ADDR_SBADDRESS0 = 7'h39;
localparam ADDR_SBDATA0 = 7'h3c;
// APB is byte-addressed, DM registers are word-addressed.
wire [6:0] dmi_regaddr = dmi_paddr[8:2];
// ----------------------------------------------------------------------------
// Hart selection
reg dmactive;
// Some fiddliness to make sure we get a single-wide zero-valued signal when
// N_HARTS == 1 (so we can use this for indexing of per-hart signals)
reg [W_HARTSEL-1:0] hartsel;
wire [W_HARTSEL-1:0] hartsel_next;
generate
if (N_HARTS > 1) begin: has_hartsel
// Only the lower 10 bits of hartsel are supported
assign hartsel_next = dmi_write && dmi_regaddr == ADDR_DMCONTROL ?
dmi_pwdata[16 +: W_HARTSEL] : hartsel;
end else begin: has_no_hartsel
assign hartsel_next = 1'b0;
end
endgenerate
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
hartsel <= {W_HARTSEL{1'b0}};
end else if (!dmactive) begin
hartsel <= {W_HARTSEL{1'b0}};
end else begin
hartsel <= hartsel_next;
end
end
// Also implement the hart array mask if there is more than one hart.
reg [N_HARTS-1:0] hart_array_mask;
reg hasel;
wire [N_HARTS-1:0] hart_array_mask_next;
wire hasel_next;
generate
if (N_HARTS > 1) begin: has_array_mask
assign hart_array_mask_next = dmi_write && dmi_regaddr == ADDR_HAWINDOW ?
dmi_pwdata[N_HARTS-1:0] : hart_array_mask;
assign hasel_next = dmi_write && dmi_regaddr == ADDR_DMCONTROL ?
dmi_pwdata[26] : hasel;
end else begin: has_no_array_mask
assign hart_array_mask_next = 1'b0;
assign hasel_next = 1'b0;
end
endgenerate
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
hart_array_mask <= {N_HARTS{1'b0}};
hasel <= 1'b0;
end else if (!dmactive) begin
hart_array_mask <= {N_HARTS{1'b0}};
hasel <= 1'b0;
end else begin
hart_array_mask <= hart_array_mask_next;
hasel <= hasel_next;
end
end
// ----------------------------------------------------------------------------
// Run/halt/reset control
// Normal read/write fields for dmcontrol (note some of these are per-hart
// fields that get rotated into dmcontrol based on the current/next hartsel).
reg [N_HARTS-1:0] dmcontrol_haltreq;
reg [N_HARTS-1:0] dmcontrol_hartreset;
reg [N_HARTS-1:0] dmcontrol_resethaltreq;
reg dmcontrol_ndmreset;
wire [N_HARTS-1:0] dmcontrol_op_mask;
generate
if (N_HARTS > 1) begin: dmcontrol_multiple_harts
// Selection is the hart selected by hartsel, *plus* the hart array mask
// if hasel is set. Note we don't need to use the "next" version of
// hart_array_mask since it can't change simultaneously with dmcontrol.
assign dmcontrol_op_mask =
(hartsel_next >= N_HARTS ? {N_HARTS{1'b0}} : {{N_HARTS-1{1'b0}}, 1'b1} << hartsel_next)
| ({N_HARTS{hasel_next}} & hart_array_mask);
end else begin: dmcontrol_single_hart
assign dmcontrol_op_mask = 1'b1;
end
endgenerate
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
dmactive <= 1'b0;
dmcontrol_ndmreset <= 1'b0;
dmcontrol_haltreq <= {N_HARTS{1'b0}};
dmcontrol_hartreset <= {N_HARTS{1'b0}};
dmcontrol_resethaltreq <= {N_HARTS{1'b0}};
end else if (!dmactive) begin
// Only dmactive is writable when !dmactive
if (dmi_write && dmi_regaddr == ADDR_DMCONTROL)
dmactive <= dmi_pwdata[0];
dmcontrol_ndmreset <= 1'b0;
dmcontrol_haltreq <= {N_HARTS{1'b0}};
dmcontrol_hartreset <= {N_HARTS{1'b0}};
dmcontrol_resethaltreq <= {N_HARTS{1'b0}};
end else if (dmi_write && dmi_regaddr == ADDR_DMCONTROL) begin
dmactive <= dmi_pwdata[0];
dmcontrol_ndmreset <= dmi_pwdata[1];
dmcontrol_haltreq <= (dmcontrol_haltreq & ~dmcontrol_op_mask) |
({N_HARTS{dmi_pwdata[31]}} & dmcontrol_op_mask);
dmcontrol_hartreset <= (dmcontrol_hartreset & ~dmcontrol_op_mask) |
({N_HARTS{dmi_pwdata[29]}} & dmcontrol_op_mask);
dmcontrol_resethaltreq <= (dmcontrol_resethaltreq
& ~({N_HARTS{dmi_pwdata[2]}} & dmcontrol_op_mask))
| ({N_HARTS{dmi_pwdata[3]}} & dmcontrol_op_mask);
end
end
assign sys_reset_req = dmcontrol_ndmreset;
assign hart_reset_req = dmcontrol_hartreset;
assign hart_req_halt = dmcontrol_haltreq;
assign hart_req_halt_on_reset = dmcontrol_resethaltreq;
reg [N_HARTS-1:0] hart_reset_done_prev;
reg [N_HARTS-1:0] dmstatus_havereset;
wire [N_HARTS-1:0] hart_available = hart_reset_done & {N_HARTS{sys_reset_done}};
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
hart_reset_done_prev <= {N_HARTS{1'b0}};
end else begin
hart_reset_done_prev <= hart_reset_done;
end
end
wire dmcontrol_ackhavereset = dmi_write && dmi_regaddr == ADDR_DMCONTROL && dmi_pwdata[28];
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
dmstatus_havereset <= {N_HARTS{1'b0}};
end else if (!dmactive) begin
dmstatus_havereset <= {N_HARTS{1'b0}};
end else begin
dmstatus_havereset <= (dmstatus_havereset | (hart_reset_done & ~hart_reset_done_prev))
& ~({N_HARTS{dmcontrol_ackhavereset}} & dmcontrol_op_mask);
end
end
reg [N_HARTS-1:0] dmstatus_resumeack;
reg [N_HARTS-1:0] dmcontrol_resumereq_sticky;
// Note: we are required to ignore resumereq when haltreq is also set, as per
// spec (odd since the host is forbidden from writing both at once anyway).
// The wording is odd, it refers only to `haltreq` which is specifically the
// write-only `dmcontrol` field, not the underlying halt request state bits.
wire dmcontrol_resumereq = dmi_write && dmi_regaddr == ADDR_DMCONTROL &&
dmi_pwdata[30] && !dmi_pwdata[31];
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
dmstatus_resumeack <= {N_HARTS{1'b0}};
dmcontrol_resumereq_sticky <= {N_HARTS{1'b0}};
end else if (!dmactive) begin
dmstatus_resumeack <= {N_HARTS{1'b0}};
dmcontrol_resumereq_sticky <= {N_HARTS{1'b0}};
end else begin
dmstatus_resumeack <= (dmstatus_resumeack
| (dmcontrol_resumereq_sticky & hart_running & hart_available))
& ~({N_HARTS{dmcontrol_resumereq}} & dmcontrol_op_mask);
dmcontrol_resumereq_sticky <= (dmcontrol_resumereq_sticky
& ~(hart_running & hart_available))
| ({N_HARTS{dmcontrol_resumereq}} & dmcontrol_op_mask);
end
end
assign hart_req_resume = dmcontrol_resumereq_sticky;
// ----------------------------------------------------------------------------
// System bus access
reg [31:0] sbaddress;
reg [31:0] sbdata;
// Update logic for address/data registers:
reg sbbusy;
reg sbautoincrement;
reg [2:0] sbaccess; // Size of the transfer
wire sbdata_write_blocked;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
sbaddress <= {32{1'b0}};
sbdata <= {32{1'b0}};
end else if (!dmactive) begin
sbaddress <= {32{1'b0}};
sbdata <= {32{1'b0}};
end else if (HAVE_SBA) begin
if (dmi_write && dmi_regaddr == ADDR_SBDATA0 && !sbdata_write_blocked) begin
// Note sbbusyerror and sberror block writes to sbdata0, as the
// write is required to have no side effects when they are set.
sbdata <= dmi_pwdata;
end else if (sbus_vld && sbus_rdy && !sbus_write && !sbus_err) begin
// Make sure the lower byte lanes see appropriately shifted data as
// long as the transfer is naturally aligned
sbdata <= sbaddress[1:0] == 2'b01 ? {sbus_rdata[31:8], sbus_rdata[15:8]} :
sbaddress[1:0] == 2'b10 ? {sbus_rdata[31:16], sbus_rdata[31:16]} :
sbaddress[1:0] == 2'b11 ? {sbus_rdata[31:8], sbus_rdata[31:24]} : sbus_rdata;
end
if (dmi_write && dmi_regaddr == ADDR_SBADDRESS0 && !sbbusy) begin
// Note sbaddress can't be written when busy, but
// sberror/sbbusyerror do not prevent writes.
sbaddress <= dmi_pwdata;
end else if (sbus_vld && sbus_rdy && !sbus_err && sbautoincrement) begin
// Note: address increments only following a successful transfer.
// Spec 0.13.2 weirdly implies address should increment following
// a sbdata0 read with sbautoincrement=1 and sbreadondata=0, but
// this seems to be a typo, fixed in later versions.
sbaddress <= sbaddress + (
sbaccess[1:0] == 2'b00 ? 32'd1 :
sbaccess[1:0] == 2'b01 ? 32'd2 : 32'd4
);
end
end
end
// Control logic:
reg sbbusyerror;
reg sbreadonaddr;
reg sbreadondata;
reg [2:0] sberror;
reg sb_current_is_write;
localparam SBERROR_OK = 3'h0;
localparam SBERROR_BADADDR = 3'h2;
localparam SBERROR_BADALIGN = 3'h3;
localparam SBERROR_BADSIZE = 3'h4;
assign sbdata_write_blocked = sbbusy || sbbusyerror || |sberror;
// Notes on behaviour of sbbusyerror: the sbbusyerror description says:
//
// "Set when the debugger attempts to read data while a read is in progress,
// or when the debugger initiates a new access while one is already in
// progress (while sbbusy is set)."
//
// However, sbaddress0 description says:
//
// "When the system bus master is busy, writes to this register will set
// sbbusyerror and dont do anything else."
//
// ...not conditioned on sbreadonaddr. Likewise the sbdata0 description says:
//
// "If the bus master is busy then accesses set sbbusyerror, and dont do
// anything else."
//
// ...not conditioned on sbreadondata. We are going to take the union of all
// the cases where the spec says we should raise an error:
wire sb_access_illegal_when_busy =
dmi_regaddr == ADDR_SBDATA0 && (dmi_read || dmi_write) ||
dmi_regaddr == ADDR_SBADDRESS0 && dmi_write;
wire sb_want_start_write = dmi_write && dmi_regaddr == ADDR_SBDATA0;
wire sb_want_start_read =
(sbreadonaddr && dmi_write && dmi_regaddr == ADDR_SBADDRESS0) ||
(sbreadondata && dmi_read && dmi_regaddr == ADDR_SBDATA0);
wire [1:0] sb_next_align = sbreadonaddr && dmi_write && dmi_regaddr == ADDR_SBADDRESS0 ?
dmi_pwdata[1:0] : sbaddress[1:0];
wire sb_badalign =
(sbaccess == 3'h1 && sb_next_align[0]) ||
(sbaccess == 3'h2 && |sb_next_align[1:0]);
wire sb_badsize = sbaccess > 3'h2;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
sbbusy <= 1'b0;
sbbusyerror <= 1'b0;
sbreadonaddr <= 1'b0;
sbreadondata <= 1'b0;
sbaccess <= 3'h0;
sbautoincrement <= 1'b0;
sberror <= 3'h0;
sb_current_is_write <= 1'b0;
end else if (!dmactive) begin
sbbusy <= 1'b0;
sbbusyerror <= 1'b0;
sbreadonaddr <= 1'b0;
sbreadondata <= 1'b0;
sbaccess <= 3'h0;
sbautoincrement <= 1'b0;
sberror <= 3'h0;
sb_current_is_write <= 1'b0;
end else if (HAVE_SBA) begin
if (dmi_write && dmi_regaddr == ADDR_SBCS) begin
// Assume a transfer is not in progress when written (per spec)
sbbusyerror <= sbbusyerror && !dmi_pwdata[22];
sbreadonaddr <= dmi_pwdata[20];
sbaccess <= dmi_pwdata[19:17];
sbautoincrement <= dmi_pwdata[16];
sbreadondata <= dmi_pwdata[15];
sberror <= sberror & ~dmi_pwdata[14:12];
end
if (sbbusy) begin
if (sb_access_illegal_when_busy) begin
sbbusyerror <= 1'b1;
end
if (sbus_vld && sbus_rdy) begin
sbbusy <= 1'b0;
if (sbus_err) begin
sberror <= SBERROR_BADADDR;
end
end
end else if ((sb_want_start_read || sb_want_start_write) && ~|sberror && !sbbusyerror) begin
if (sb_badsize) begin
sberror <= SBERROR_BADSIZE;
end else if (sb_badalign) begin
sberror <= SBERROR_BADALIGN;
end else begin
sbbusy <= 1'b1;
sb_current_is_write <= sb_want_start_write;
end
end
end
end
assign sbus_addr = sbaddress;
assign sbus_write = sb_current_is_write;
assign sbus_size = sbaccess[1:0];
assign sbus_vld = sbbusy;
// Replicate byte lanes to handle naturally-aligned cases.
assign sbus_wdata = sbaccess[1:0] == 2'b00 ? {4{sbdata[7:0]}} :
sbaccess[1:0] == 2'b01 ? {2{sbdata[15:0]}} : sbdata;
// ----------------------------------------------------------------------------
// Abstract command data registers
wire abstractcs_busy;
// The same data0 register is aliased as a CSR on all harts connected to this
// DM. Cores may read data0 as a CSR when in debug mode, and may write it when:
//
// - That core is in debug mode, and...
// - We are currently executing an abstract command on that core
//
// The DM can also read/write data0 at all times.
reg [XLEN-1:0] abstract_data0;
assign hart_data0_rdata = {N_HARTS{abstract_data0}};
always @ (posedge clk or negedge rst_n) begin: update_hart_data0
reg signed [31:0] i;
if (!rst_n) begin
abstract_data0 <= {XLEN{1'b0}};
end else if (!dmactive) begin
abstract_data0 <= {XLEN{1'b0}};
end else if (dmi_write && dmi_regaddr == ADDR_DATA0) begin
abstract_data0 <= dmi_pwdata;
end else begin
for (i = 0; i < N_HARTS; i = i + 1) begin
if (hartsel == i[W_HARTSEL-1:0] && hart_data0_wen[i] && hart_halted[i] && abstractcs_busy)
abstract_data0 <= hart_data0_wdata[i * XLEN +: XLEN];
end
end
end
reg [XLEN-1:0] progbuf0;
reg [XLEN-1:0] progbuf1;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
progbuf0 <= {XLEN{1'b0}};
progbuf1 <= {XLEN{1'b0}};
end else if (!dmactive) begin
progbuf0 <= {XLEN{1'b0}};
progbuf1 <= {XLEN{1'b0}};
end else if (dmi_write && !abstractcs_busy) begin
if (dmi_regaddr == ADDR_PROGBUF0)
progbuf0 <= dmi_pwdata;
if (dmi_regaddr == ADDR_PROGBUF1)
progbuf1 <= dmi_pwdata;
end
end
reg abstractauto_autoexecdata;
reg [1:0] abstractauto_autoexecprogbuf;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
abstractauto_autoexecdata <= 1'b0;
abstractauto_autoexecprogbuf <= 2'b00;
end else if (!dmactive) begin
abstractauto_autoexecdata <= 1'b0;
abstractauto_autoexecprogbuf <= 2'b00;
end else if (dmi_write && dmi_regaddr == ADDR_ABSTRACTAUTO) begin
abstractauto_autoexecdata <= dmi_pwdata[0];
abstractauto_autoexecprogbuf <= dmi_pwdata[17:16];
end
end
// ----------------------------------------------------------------------------
// Abstract command state machine
localparam W_STATE = 4;
localparam S_IDLE = 4'd0;
localparam S_ISSUE_REGREAD = 4'd1;
localparam S_ISSUE_REGWRITE = 4'd2;
localparam S_ISSUE_REGEBREAK = 4'd3;
localparam S_WAIT_REGEBREAK = 4'd4;
localparam S_ISSUE_PROGBUF0 = 4'd5;
localparam S_ISSUE_PROGBUF1 = 4'd6;
localparam S_ISSUE_IMPEBREAK = 4'd7;
localparam S_WAIT_IMPEBREAK = 4'd8;
localparam CMDERR_OK = 3'h0;
localparam CMDERR_BUSY = 3'h1;
localparam CMDERR_UNSUPPORTED = 3'h2;
localparam CMDERR_EXCEPTION = 3'h3;
localparam CMDERR_HALTRESUME = 3'h4;
reg [2:0] abstractcs_cmderr;
reg [2:0] abstractcs_cmderr_nxt;
reg [W_STATE-1:0] acmd_state;
reg [W_STATE-1:0] acmd_state_nxt;
assign abstractcs_busy = acmd_state != S_IDLE;
wire start_abstract_cmd = abstractcs_cmderr == CMDERR_OK && !abstractcs_busy && (
(dmi_write && dmi_regaddr == ADDR_COMMAND) ||
((dmi_write || dmi_read) && abstractauto_autoexecdata && dmi_regaddr == ADDR_DATA0) ||
((dmi_write || dmi_read) && abstractauto_autoexecprogbuf[0] && dmi_regaddr == ADDR_PROGBUF0) ||
((dmi_write || dmi_read) && abstractauto_autoexecprogbuf[1] && dmi_regaddr == ADDR_PROGBUF1)
);
wire dmi_access_illegal_when_busy =
(dmi_write && (
dmi_regaddr == ADDR_ABSTRACTCS || dmi_regaddr == ADDR_COMMAND || dmi_regaddr == ADDR_ABSTRACTAUTO ||
dmi_regaddr == ADDR_DATA0 || dmi_regaddr == ADDR_PROGBUF0 || dmi_regaddr == ADDR_PROGBUF1)) ||
(dmi_read && (
dmi_regaddr == ADDR_DATA0 || dmi_regaddr == ADDR_PROGBUF0 || dmi_regaddr == ADDR_PROGBUF1));
// Decode what acmd may be triggered on this cycle, and whether it is
// supported -- command source may be a registered version of most recent
// command (if abstractauto is used) or a fresh command off the bus. We don't
// register the entire write data; repeats of unsupported commands are
// detected by just registering that the last written command was
// unsupported.
wire acmd_new = dmi_write && dmi_regaddr == ADDR_COMMAND;
wire acmd_new_postexec = dmi_pwdata[18];
wire acmd_new_transfer = dmi_pwdata[17];
wire acmd_new_write = dmi_pwdata[16];
wire [4:0] acmd_new_regno = dmi_pwdata[4:0];
// Note: regno and aarsize are permitted to have otherwise-invalid values if
// the transfer flag is not set.
wire acmd_new_unsupported =
(dmi_pwdata[31:24] != 8'h00 ) || // Only Access Register command supported
(dmi_pwdata[22:20] != 3'h2 && acmd_new_transfer) || // Must be 32 bits in size
(dmi_pwdata[19] ) || // aarpostincrement not supported
(dmi_pwdata[15:12] != 4'h1 && acmd_new_transfer) || // Only core register access supported
(dmi_pwdata[11:5] != 7'h0 && acmd_new_transfer); // Only GPRs supported
reg acmd_prev_postexec;
reg acmd_prev_transfer;
reg acmd_prev_write;
reg [4:0] acmd_prev_regno;
reg acmd_prev_unsupported;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
acmd_prev_postexec <= 1'b0;
acmd_prev_transfer <= 1'b0;
acmd_prev_write <= 1'b0;
acmd_prev_regno <= 5'h0;
acmd_prev_unsupported <= 1'b1;
end else if (!dmactive) begin
acmd_prev_postexec <= 1'b0;
acmd_prev_transfer <= 1'b0;
acmd_prev_write <= 1'b0;
acmd_prev_regno <= 5'h0;
acmd_prev_unsupported <= 1'b1;
end else if (start_abstract_cmd && acmd_new) begin
acmd_prev_postexec <= acmd_new_postexec;
acmd_prev_transfer <= acmd_new_transfer;
acmd_prev_write <= acmd_new_write;
acmd_prev_regno <= acmd_new_regno;
acmd_prev_unsupported <= acmd_new_unsupported;
end
end
wire acmd_postexec = acmd_new ? acmd_new_postexec : acmd_prev_postexec ;
wire acmd_transfer = acmd_new ? acmd_new_transfer : acmd_prev_transfer ;
wire acmd_write = acmd_new ? acmd_new_write : acmd_prev_write ;
wire [4:0] acmd_regno = acmd_new ? acmd_new_regno : acmd_prev_regno ;
wire acmd_unsupported = acmd_new ? acmd_new_unsupported : acmd_prev_unsupported;
always @ (*) begin
// Default: no state change
acmd_state_nxt = acmd_state;
abstractcs_cmderr_nxt = abstractcs_cmderr;
if (dmi_write && dmi_regaddr == ADDR_ABSTRACTCS && !abstractcs_busy)
abstractcs_cmderr_nxt = abstractcs_cmderr & ~dmi_pwdata[10:8];
if (abstractcs_cmderr == CMDERR_OK && abstractcs_busy && dmi_access_illegal_when_busy)
abstractcs_cmderr_nxt = CMDERR_BUSY;
if (acmd_state != S_IDLE && hart_instr_caught_exception[hartsel])
abstractcs_cmderr_nxt = CMDERR_EXCEPTION;
case (acmd_state)
S_IDLE: begin
if (start_abstract_cmd) begin
if (!hart_halted[hartsel] || !hart_available[hartsel]) begin
abstractcs_cmderr_nxt = CMDERR_HALTRESUME;
end else if (acmd_unsupported) begin
abstractcs_cmderr_nxt = CMDERR_UNSUPPORTED;
end else begin
if (acmd_transfer && acmd_write)
acmd_state_nxt = S_ISSUE_REGWRITE;
else if (acmd_transfer && !acmd_write)
acmd_state_nxt = S_ISSUE_REGREAD;
else if (acmd_postexec)
acmd_state_nxt = S_ISSUE_PROGBUF0;
else
acmd_state_nxt = S_IDLE;
end
end
end
S_ISSUE_REGREAD: begin
if (hart_instr_data_rdy[hartsel])
acmd_state_nxt = S_ISSUE_REGEBREAK;
end
S_ISSUE_REGWRITE: begin
if (hart_instr_data_rdy[hartsel])
acmd_state_nxt = S_ISSUE_REGEBREAK;
end
S_ISSUE_REGEBREAK: begin
if (hart_instr_data_rdy[hartsel])
acmd_state_nxt = S_WAIT_REGEBREAK;
end
S_WAIT_REGEBREAK: begin
if (hart_instr_caught_ebreak[hartsel]) begin
if (acmd_prev_postexec)
acmd_state_nxt = S_ISSUE_PROGBUF0;
else
acmd_state_nxt = S_IDLE;
end
end
S_ISSUE_PROGBUF0: begin
if (hart_instr_data_rdy[hartsel])
acmd_state_nxt = S_ISSUE_PROGBUF1;
end
S_ISSUE_PROGBUF1: begin
if (hart_instr_caught_exception[hartsel] || hart_instr_caught_ebreak[hartsel]) begin
acmd_state_nxt = S_IDLE;
end else if (hart_instr_data_rdy[hartsel]) begin
acmd_state_nxt = S_ISSUE_IMPEBREAK;
end
end
S_ISSUE_IMPEBREAK: begin
if (hart_instr_caught_exception[hartsel] || hart_instr_caught_ebreak[hartsel]) begin
acmd_state_nxt = S_IDLE;
end else if (hart_instr_data_rdy[hartsel]) begin
acmd_state_nxt = S_WAIT_IMPEBREAK;
end
end
S_WAIT_IMPEBREAK: begin
if (hart_instr_caught_exception[hartsel] || hart_instr_caught_ebreak[hartsel]) begin
acmd_state_nxt = S_IDLE;
end
end
default: begin
// Unreachable
end
endcase
end
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
abstractcs_cmderr <= CMDERR_OK;
acmd_state <= S_IDLE;
end else if (!dmactive) begin
abstractcs_cmderr <= CMDERR_OK;
acmd_state <= S_IDLE;
end else begin
abstractcs_cmderr <= abstractcs_cmderr_nxt;
acmd_state <= acmd_state_nxt;
end
end
wire [N_HARTS-1:0] hart_instr_data_vld_nxt = {{N_HARTS-1{1'b0}},
acmd_state_nxt == S_ISSUE_REGREAD || acmd_state_nxt == S_ISSUE_REGWRITE || acmd_state_nxt == S_ISSUE_REGEBREAK ||
acmd_state_nxt == S_ISSUE_PROGBUF0 || acmd_state_nxt == S_ISSUE_PROGBUF1 || acmd_state_nxt == S_ISSUE_IMPEBREAK
} << hartsel;
wire [31:0] hart_instr_data_nxt =
acmd_state_nxt == S_ISSUE_REGWRITE ? 32'hbff02073 | {20'd0, acmd_regno, 7'd0} : // csrr xx, dmdata0
acmd_state_nxt == S_ISSUE_REGREAD ? 32'hbff01073 | {12'd0, acmd_regno, 15'd0} : // csrw dmdata0, xx
acmd_state_nxt == S_ISSUE_PROGBUF0 ? progbuf0 :
acmd_state_nxt == S_ISSUE_PROGBUF1 ? progbuf1 :
32'h00100073; // ebreak
reg [31:0] hart_instr_data_reg;
assign hart_instr_data = {N_HARTS{hart_instr_data_reg}};
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
hart_instr_data_vld <= 1'b0;
hart_instr_data_reg <= 32'h00000000;
end else begin
hart_instr_data_vld <= hart_instr_data_vld_nxt;
if (hart_instr_data_vld_nxt) begin
hart_instr_data_reg <= hart_instr_data_nxt;
end
end
end
// ----------------------------------------------------------------------------
// Status helper functions
function status_any;
input [N_HARTS-1:0] status_mask;
begin
status_any = status_mask[hartsel] || (hasel && |(status_mask & hart_array_mask));
end
endfunction
function status_all;
input [N_HARTS-1:0] status_mask;
begin
status_all = status_mask[hartsel] && (!hasel || ~|(~status_mask & hart_array_mask));
end
endfunction
function [1:0] status_all_any;
input [N_HARTS-1:0] status_mask;
begin
status_all_any = {
status_all(status_mask),
status_any(status_mask)
};
end
endfunction
// ----------------------------------------------------------------------------
// DMI read data mux
always @ (*) begin
case (dmi_regaddr)
ADDR_DATA0: dmi_prdata = abstract_data0;
ADDR_DMCONTROL: dmi_prdata = {
1'b0, // haltreq is a W-only field
1'b0, // resumereq is a W1 field
status_any(dmcontrol_hartreset),
1'b0, // ackhavereset is a W1 field
1'b0, // reserved
hasel,
{{10-W_HARTSEL{1'b0}}, hartsel}, // hartsello
10'h0, // hartselhi
2'h0, // reserved
2'h0, // set/clrresethaltreq are W1 fields
dmcontrol_ndmreset,
dmactive
};
ADDR_DMSTATUS: dmi_prdata = {
9'h0, // reserved
1'b1, // impebreak = 1
2'h0, // reserved
status_all_any(dmstatus_havereset), // allhavereset, anyhavereset
status_all_any(dmstatus_resumeack), // allresumeack, anyresumeack
hartsel >= N_HARTS && !(hasel && |hart_array_mask), // allnonexistent
hartsel >= N_HARTS, // anynonexistent
status_all_any(~hart_available), // allunavail, anyunavail
status_all_any(hart_running & hart_available), // allrunning, anyrunning
status_all_any(hart_halted & hart_available), // allhalted, anyhalted
1'b1, // authenticated
1'b0, // authbusy
1'b1, // hasresethaltreq = 1 (we do support it)
1'b0, // confstrptrvalid
4'd2 // version = 2: RISC-V debug spec 0.13.2
};
ADDR_HARTINFO: dmi_prdata = {
8'h0, // reserved
4'h0, // nscratch = 0
3'h0, // reserved
1'b0, // dataccess = 0, data0 is mapped to each hart's CSR space
4'h1, // datasize = 1, a single data CSR (data0) is available
12'hbff // dataaddr, placed at the top of the M-custom space since
// the spec doesn't reserve a location for it.
};
ADDR_HALTSUM0: dmi_prdata = {
{XLEN - N_HARTS{1'b0}},
hart_halted & hart_available
};
ADDR_HALTSUM1: dmi_prdata = {
{XLEN - 1{1'b0}},
|(hart_halted & hart_available)
};
ADDR_HAWINDOWSEL: dmi_prdata = 32'h00000000;
ADDR_HAWINDOW: dmi_prdata = {
{32-N_HARTS{1'b0}},
hart_array_mask
};
ADDR_ABSTRACTCS: dmi_prdata = {
3'h0, // reserved
5'd2, // progbufsize = 2
11'h0, // reserved
abstractcs_busy,
1'b0,
abstractcs_cmderr,
4'h0,
4'd1 // datacount = 1
};
ADDR_ABSTRACTAUTO: dmi_prdata = {
14'h0,
abstractauto_autoexecprogbuf, // only progbuf0,1 present
15'h0,
abstractauto_autoexecdata // only data0 present
};
ADDR_SBCS: dmi_prdata = {
3'h1, // version = 1
6'h00,
sbbusyerror,
sbbusy,
sbreadonaddr,
sbaccess,
sbautoincrement,
sbreadondata,
sberror,
7'h20, // sbasize = 32
5'b00111 // 8, 16, 32-bit transfers supported
} & {32{|HAVE_SBA}};
ADDR_SBDATA0: dmi_prdata = sbdata & {32{|HAVE_SBA}};
ADDR_SBADDRESS0: dmi_prdata = sbaddress & {32{|HAVE_SBA}};
ADDR_CONFSTRPTR0: dmi_prdata = 32'h4c296328;
ADDR_CONFSTRPTR1: dmi_prdata = 32'h20656b75;
ADDR_CONFSTRPTR2: dmi_prdata = 32'h6e657257;
ADDR_CONFSTRPTR3: dmi_prdata = 32'h31322720;
ADDR_NEXTDM: dmi_prdata = NEXT_DM_ADDR;
ADDR_PROGBUF0: dmi_prdata = progbuf0;
ADDR_PROGBUF1: dmi_prdata = progbuf1;
default: dmi_prdata = {XLEN{1'b0}};
endcase
end
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+74
View File
@@ -0,0 +1,74 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Standalone bus shim for connecting the DM's System Bus Access to AHB
`default_nettype none
module hazard3_sbus_to_ahb #(
parameter W_ADDR = 32,
parameter W_DATA = 32
) (
input wire clk,
input wire rst_n,
input wire [W_ADDR-1:0] sbus_addr,
input wire sbus_write,
input wire [1:0] sbus_size,
input wire sbus_vld,
output wire sbus_rdy,
output wire sbus_err,
input wire [W_DATA-1:0] sbus_wdata,
output wire [W_DATA-1:0] sbus_rdata,
output wire [W_ADDR-1:0] ahblm_haddr,
output wire ahblm_hwrite,
output wire [1:0] ahblm_htrans,
output wire [2:0] ahblm_hsize,
output wire [2:0] ahblm_hburst,
output wire [3:0] ahblm_hprot,
output wire ahblm_hmastlock,
input wire ahblm_hready,
input wire ahblm_hresp,
output wire [W_DATA-1:0] ahblm_hwdata,
input wire [W_DATA-1:0] ahblm_hrdata
);
// Most signals are simple tie-throughs
assign ahblm_haddr = sbus_addr;
assign ahblm_hwrite = sbus_write;
assign ahblm_hsize = {1'b0, sbus_size};
assign ahblm_hwdata = sbus_wdata;
// HPROT = noncacheable nonbufferable privileged data access:
assign ahblm_hprot = 4'b0011;
assign ahblm_hmastlock = 1'b0;
assign ahblm_hburst = 3'h0;
assign sbus_err = ahblm_hresp;
assign sbus_rdata = ahblm_hrdata;
// Handshaking
reg dph_active;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
dph_active <= 1'b0;
end else if (ahblm_hready) begin
dph_active <= ahblm_htrans[1];
end
end
assign ahblm_htrans = sbus_vld && !dph_active ? 2'b10 : 2'b00;
assign sbus_rdy = ahblm_hready && dph_active;
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+6
View File
@@ -0,0 +1,6 @@
file hazard3_ecp5_jtag_dtm.v
file hazard3_jtag_dtm_core.v
file ../cdc/hazard3_apb_async_bridge.v
file ../cdc/hazard3_reset_sync.v
file ../cdc/hazard3_sync_1bit.v
+194
View File
@@ -0,0 +1,194 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// The ECP5 JTAGG primitive (yes that is the correct spelling) allows you to
// add two custom DRs to the FPGA's chip TAP, selected using the 8-bit ER1
// (0x32) and ER2 (0x38) instructions.
//
// Brian Swetland pointed out on Twitter that the standard RISC-V JTAG-DTM
// only uses two DRs (DTMCS and DMI), besides the standard IDCODE and BYPASS
// which are provided already by the ECP5 TAP. This file instantiates the
// guts of Hazard3's standard JTAG-DTM and connects the DTMCS and DMI
// registers to the JTAGG primitive's ER1/ER2 DRs.
//
// The exciting part is that upstream OpenOCD already allows you to set the IR
// length *and* set custom DTMCS/DMI IR values for RISC-V JTAG DTMs. This
// means with the right config file, you can access a debug module hung from
// the ECP5 TAP in this fashion using only upstream OpenOCD and gdb.
`default_nettype none
module hazard3_ecp5_jtag_dtm #(
parameter DTMCS_IDLE_HINT = 3'd4,
parameter W_PADDR = 9,
parameter ABITS = W_PADDR - 2 // do not modify
) (
// This is synchronous to TCK and asserted for one TCK cycle only
output wire dmihardreset_req,
// Bus clock + reset for Debug Module Interface
input wire clk_dmi,
input wire rst_n_dmi,
// Debug Module Interface (APB)
output wire dmi_psel,
output wire dmi_penable,
output wire dmi_pwrite,
output wire [W_PADDR-1:0] dmi_paddr,
output wire [31:0] dmi_pwdata,
input wire [31:0] dmi_prdata,
input wire dmi_pready,
input wire dmi_pslverr
);
// Signals to/from the ECP5 TAP
wire jtdo2;
wire jtdo1;
wire jtdi;
wire jtck_posedge_dont_use;
wire jshift;
wire jupdate;
wire jrst_n;
wire jce2;
wire jce1;
JTAGG jtag_u (
.JTDO2 (jtdo2),
.JTDO1 (jtdo1),
.JTDI (jtdi),
.JTCK (jtck_posedge_dont_use),
.JRTI2 (/* unused */),
.JRTI1 (/* unused */),
.JSHIFT (jshift),
.JUPDATE (jupdate),
.JRSTN (jrst_n),
.JCE2 (jce2),
.JCE1 (jce1)
);
// JTAGG primitive asserts its signals synchronously to JTCK's posedge, but
// you get weird and inconsistent results if you try to consume them
// synchronously on JTCK's posedge, possibly due to a lack of hold
// constraints in nextpnr.
//
// A quick hack is to move the sampling onto the negedge of the clock. This
// then creates more problems because we would be running our shift logic on
// a different edge from the control + CDC logic in the DTM core.
//
// So, even worse hack, move all our JTAG-domain logic onto the negedge
// (or near enough) by inverting the clock.
wire jtck = !jtck_posedge_dont_use;
localparam W_DR_SHIFT = ABITS + 32 + 2;
reg core_dr_wen;
reg core_dr_ren;
reg core_dr_sel_dmi_ndtmcs;
reg dr_shift_en;
wire [W_DR_SHIFT-1:0] core_dr_wdata;
wire [W_DR_SHIFT-1:0] core_dr_rdata;
// Decode our shift controls from the interesting ECP5 ones, and re-register
// onto JTCK negedge (our posedge). Note without re-registering we observe
// them a half-cycle (effectively one cycle) too early. This is another
// consequence of the stupid JTDI thing
always @ (posedge jtck or negedge jrst_n) begin
if (!jrst_n) begin
core_dr_sel_dmi_ndtmcs <= 1'b0;
core_dr_wen <= 1'b0;
core_dr_ren <= 1'b0;
dr_shift_en <= 1'b0;
end else begin
if (jce1 || jce2)
core_dr_sel_dmi_ndtmcs <= jce2;
core_dr_ren <= (jce1 || jce2) && !jshift;
core_dr_wen <= jupdate;
dr_shift_en <= jshift;
end
end
reg [W_DR_SHIFT-1:0] dr_shift;
assign core_dr_wdata = dr_shift;
always @ (posedge jtck or negedge jrst_n) begin
if (!jrst_n) begin
dr_shift <= {W_DR_SHIFT{1'b0}};
end else if (core_dr_ren) begin
dr_shift <= core_dr_rdata;
end else if (dr_shift_en) begin
dr_shift <= {jtdi, dr_shift[W_DR_SHIFT-1:1]};
if (!core_dr_sel_dmi_ndtmcs)
dr_shift[31] <= jtdi;
end
end
// Not documented on ECP5: as well as the posedge flop on JTDI, the ECP5 puts
// a negedge flop on JTDO1, JTDO2. (Conjecture based on dicking around with a
// logic analyser.) To get JTDOx to appear with the same timing as our shifter
// LSB (which we update on every JTCK negedge) we:
//
// - Register the LSB of the *next* value of dr_shift on the JTCK posedge, so
// half a cycle earlier than the actual dr_shift update
//
// - This then gets re-registered with the pointless JTDO negedge flops, so
// that it appears with the same timing as our DR shifter update.
reg dr_shift_next_halfcycle;
always @ (negedge jtck or negedge jrst_n) begin
if (!jrst_n) begin
dr_shift_next_halfcycle <= 1'b0;
end else begin
dr_shift_next_halfcycle <=
core_dr_ren ? core_dr_rdata[0] :
dr_shift_en ? dr_shift[1] : dr_shift[0];
end
end
// We have only a single shifter for the ER1 and ER2 chains, so these are tied
// together:
assign jtdo1 = dr_shift_next_halfcycle;
assign jtdo2 = dr_shift_next_halfcycle;
// The actual DTM is in here:
hazard3_jtag_dtm_core #(
.DTMCS_IDLE_HINT (DTMCS_IDLE_HINT),
.W_ADDR (ABITS)
) inst_hazard3_jtag_dtm_core (
.tck (jtck),
.trst_n (jrst_n),
.clk_dmi (clk_dmi),
.rst_n_dmi (rst_n_dmi),
.dr_wen (core_dr_wen),
.dr_ren (core_dr_ren),
.dr_sel_dmi_ndtmcs (core_dr_sel_dmi_ndtmcs),
.dr_wdata (core_dr_wdata),
.dr_rdata (core_dr_rdata),
.dmihardreset_req (dmihardreset_req),
.dmi_psel (dmi_psel),
.dmi_penable (dmi_penable),
.dmi_pwrite (dmi_pwrite),
.dmi_paddr (dmi_paddr[W_PADDR-1:2]),
.dmi_pwdata (dmi_pwdata),
.dmi_prdata (dmi_prdata),
.dmi_pready (dmi_pready),
.dmi_pslverr (dmi_pslverr)
);
assign dmi_paddr[1:0] = 2'b00;
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+6
View File
@@ -0,0 +1,6 @@
file hazard3_jtag_dtm.v
file hazard3_jtag_dtm_core.v
file ../cdc/hazard3_apb_async_bridge.v
file ../cdc/hazard3_reset_sync.v
file ../cdc/hazard3_sync_1bit.v
+210
View File
@@ -0,0 +1,210 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Implementation of standard RISC-V JTAG-DTM with an APB Debug Module
// Interface. The TAP itself is clocked directly by JTAG TCK; a clock
// crossing is instantiated internally between the TCK domain and the DMI bus
// clock domain.
`default_nettype none
module hazard3_jtag_dtm #(
parameter IDCODE = 32'h0000_0001,
parameter DTMCS_IDLE_HINT = 3'd4,
parameter W_PADDR = 9,
parameter ABITS = W_PADDR - 2 // do not modify
) (
// Standard JTAG signals -- the JTAG hardware is clocked directly by TCK.
input wire tck,
input wire trst_n,
input wire tms,
input wire tdi,
output reg tdo,
// This is synchronous to TCK and asserted for one TCK cycle only
output wire dmihardreset_req,
// Bus clock + reset for Debug Module Interface
input wire clk_dmi,
input wire rst_n_dmi,
// Debug Module Interface (APB)
output wire dmi_psel,
output wire dmi_penable,
output wire dmi_pwrite,
output wire [W_PADDR-1:0] dmi_paddr,
output wire [31:0] dmi_pwdata,
input wire [31:0] dmi_prdata,
input wire dmi_pready,
input wire dmi_pslverr
);
// ----------------------------------------------------------------------------
// TAP state machine
reg [3:0] tap_state;
localparam S_RESET = 4'd0;
localparam S_RUN_IDLE = 4'd1;
localparam S_SELECT_DR = 4'd2;
localparam S_CAPTURE_DR = 4'd3;
localparam S_SHIFT_DR = 4'd4;
localparam S_EXIT1_DR = 4'd5;
localparam S_PAUSE_DR = 4'd6;
localparam S_EXIT2_DR = 4'd7;
localparam S_UPDATE_DR = 4'd8;
localparam S_SELECT_IR = 4'd9;
localparam S_CAPTURE_IR = 4'd10;
localparam S_SHIFT_IR = 4'd11;
localparam S_EXIT1_IR = 4'd12;
localparam S_PAUSE_IR = 4'd13;
localparam S_EXIT2_IR = 4'd14;
localparam S_UPDATE_IR = 4'd15;
always @ (posedge tck or negedge trst_n) begin
if (!trst_n) begin
tap_state <= S_RESET;
end else case(tap_state)
S_RESET : tap_state <= tms ? S_RESET : S_RUN_IDLE ;
S_RUN_IDLE : tap_state <= tms ? S_SELECT_DR : S_RUN_IDLE ;
S_SELECT_DR : tap_state <= tms ? S_SELECT_IR : S_CAPTURE_DR;
S_CAPTURE_DR : tap_state <= tms ? S_EXIT1_DR : S_SHIFT_DR ;
S_SHIFT_DR : tap_state <= tms ? S_EXIT1_DR : S_SHIFT_DR ;
S_EXIT1_DR : tap_state <= tms ? S_UPDATE_DR : S_PAUSE_DR ;
S_PAUSE_DR : tap_state <= tms ? S_EXIT2_DR : S_PAUSE_DR ;
S_EXIT2_DR : tap_state <= tms ? S_UPDATE_DR : S_SHIFT_DR ;
S_UPDATE_DR : tap_state <= tms ? S_SELECT_DR : S_RUN_IDLE ;
S_SELECT_IR : tap_state <= tms ? S_RESET : S_CAPTURE_IR;
S_CAPTURE_IR : tap_state <= tms ? S_EXIT1_IR : S_SHIFT_IR ;
S_SHIFT_IR : tap_state <= tms ? S_EXIT1_IR : S_SHIFT_IR ;
S_EXIT1_IR : tap_state <= tms ? S_UPDATE_IR : S_PAUSE_IR ;
S_PAUSE_IR : tap_state <= tms ? S_EXIT2_IR : S_PAUSE_IR ;
S_EXIT2_IR : tap_state <= tms ? S_UPDATE_IR : S_SHIFT_IR ;
S_UPDATE_IR : tap_state <= tms ? S_SELECT_DR : S_RUN_IDLE ;
endcase
end
// ----------------------------------------------------------------------------
// Instruction register
localparam W_IR = 5;
// All other encodings behave as BYPASS:
localparam IR_IDCODE = 5'h01;
localparam IR_DTMCS = 5'h10;
localparam IR_DMI = 5'h11;
reg [W_IR-1:0] ir_shift;
reg [W_IR-1:0] ir;
always @ (posedge tck or negedge trst_n) begin
if (!trst_n) begin
ir_shift <= {W_IR{1'b0}};
ir <= IR_IDCODE;
end else if (tap_state == S_RESET) begin
ir_shift <= {W_IR{1'b0}};
ir <= IR_IDCODE;
end else if (tap_state == S_CAPTURE_IR) begin
ir_shift <= ir;
end else if (tap_state == S_SHIFT_IR) begin
ir_shift <= {tdi, ir_shift[W_IR-1:1]};
end else if (tap_state == S_UPDATE_IR) begin
ir <= ir_shift;
end
end
// ----------------------------------------------------------------------------
// Data registers
// Shift register is sized to largest DR, which is DMI:
// {addr[7:0], data[31:0], op[1:0]}
localparam W_DR_SHIFT = ABITS + 32 + 2;
reg [W_DR_SHIFT-1:0] dr_shift;
// Signals to/from the DTM core, which implements the DTMCS and DMI registers
wire core_dr_wen;
wire core_dr_ren;
wire core_dr_sel_dmi_ndtmcs;
wire [W_DR_SHIFT-1:0] core_dr_wdata;
wire [W_DR_SHIFT-1:0] core_dr_rdata;
always @ (posedge tck or negedge trst_n) begin
if (!trst_n) begin
dr_shift <= {W_DR_SHIFT{1'b0}};
end else if (tap_state == S_SHIFT_DR) begin
dr_shift <= {tdi, dr_shift[W_DR_SHIFT-1:1]};
// Shorten DR shift chain according to IR
if (ir == IR_DMI)
dr_shift[W_DR_SHIFT - 1] <= tdi;
else if (ir == IR_IDCODE || ir == IR_DTMCS)
dr_shift[31] <= tdi;
else // BYPASS
dr_shift[0] <= tdi;
end else if (tap_state == S_CAPTURE_DR) begin
if (ir == IR_DMI || ir == IR_DTMCS) begin
dr_shift <= core_dr_rdata;
end else if (ir == IR_IDCODE) begin
dr_shift <= {{W_DR_SHIFT-32{1'b0}}, IDCODE};
end else begin // BYPASS
dr_shift <= {W_DR_SHIFT{1'b0}};
end
end
end
// Must retime shift data onto negedge before presenting on TDO
always @ (negedge tck or negedge trst_n) begin
if (!trst_n) begin
tdo <= 1'b0;
end else begin
tdo <= tap_state == S_SHIFT_IR ? ir_shift[0] :
tap_state == S_SHIFT_DR ? dr_shift[0] : 1'b0;
end
end
// ----------------------------------------------------------------------------
// Core logic and bus interface
assign core_dr_sel_dmi_ndtmcs = ir == IR_DMI;
assign core_dr_wen = (ir == IR_DMI || ir == IR_DTMCS) && tap_state == S_UPDATE_DR;
assign core_dr_ren = (ir == IR_DMI || ir == IR_DTMCS) && tap_state == S_CAPTURE_DR;
assign core_dr_wdata = dr_shift;
hazard3_jtag_dtm_core #(
.DTMCS_IDLE_HINT (DTMCS_IDLE_HINT),
.W_ADDR (ABITS)
) dtm_core (
.tck (tck),
.trst_n (trst_n),
.clk_dmi (clk_dmi),
.rst_n_dmi (rst_n_dmi),
.dmihardreset_req (dmihardreset_req),
.dr_wen (core_dr_wen),
.dr_ren (core_dr_ren),
.dr_sel_dmi_ndtmcs (core_dr_sel_dmi_ndtmcs),
.dr_wdata (core_dr_wdata),
.dr_rdata (core_dr_rdata),
.dmi_psel (dmi_psel),
.dmi_penable (dmi_penable),
.dmi_pwrite (dmi_pwrite),
.dmi_paddr (dmi_paddr[W_PADDR-1:2]),
.dmi_pwdata (dmi_pwdata),
.dmi_prdata (dmi_prdata),
.dmi_pready (dmi_pready),
.dmi_pslverr (dmi_pslverr)
);
assign dmi_paddr[1:0] = 2'b00;
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+185
View File
@@ -0,0 +1,185 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// DTMCS + DMI control logic, bus interface and bus clock domain crossing for
// a standard RISC-V APB JTAG-DTM. Essentially everything apart from the
// actual TAP controller, IR and shift registers. Instantiated by
// hazard3_jtag_dtm.v.
//
// This core logic can be reused and connected to some other serial transport
// or, for example, the ECP5 JTAGG primitive (see hazard5_ecp5_jtag_dtm.v)
`default_nettype none
module hazard3_jtag_dtm_core #(
parameter DTMCS_IDLE_HINT = 3'd4,
parameter W_ADDR = 8,
parameter W_DR_SHIFT = W_ADDR + 32 + 2 // do not modify
) (
input wire tck,
input wire trst_n,
input wire clk_dmi,
input wire rst_n_dmi,
// DR capture/update (read/write) signals
input wire dr_wen,
input wire dr_ren,
input wire dr_sel_dmi_ndtmcs,
input wire [W_DR_SHIFT-1:0] dr_wdata,
output wire [W_DR_SHIFT-1:0] dr_rdata,
// This is synchronous to TCK and asserted for one TCK cycle only
output reg dmihardreset_req,
// Debug Module Interface (APB)
output wire dmi_psel,
output wire dmi_penable,
output wire dmi_pwrite,
output wire [W_ADDR-1:0] dmi_paddr,
output wire [31:0] dmi_pwdata,
input wire [31:0] dmi_prdata,
input wire dmi_pready,
input wire dmi_pslverr
);
wire write_dmi = dr_wen && dr_sel_dmi_ndtmcs;
wire write_dtmcs = dr_wen && !dr_sel_dmi_ndtmcs;
wire read_dmi = dr_ren && dr_sel_dmi_ndtmcs;
// ----------------------------------------------------------------------------
// DMI bus adapter
reg [1:0] dmi_cmderr;
reg dmi_busy;
// DTM-domain bus, connected to a matching DM-domain bus via an APB crossing:
wire dtm_psel;
wire dtm_penable;
wire dtm_pwrite;
wire [W_ADDR-1:0] dtm_paddr;
wire [31:0] dtm_pwdata;
wire [31:0] dtm_prdata;
wire dtm_pready;
wire dtm_pslverr;
// We are relying on some particular features of our APB clock crossing here
// to save some registers:
//
// - The transfer is launched immediately when psel is seen, no need to
// actually assert an access phase (as the standard allows the CDC to
// assume that access immediately follows setup) and no need to maintain
// pwrite/paddr/pwdata valid after the setup phase
//
// - prdata/pslverr remain valid after the transfer completes, until the next
// transfer completes
//
// These allow us to connect the upstream side of the CDC directly to our DR
// shifter without any sample/hold registers in between.
// psel is only pulsed for one cycle, penable is not asserted.
assign dtm_psel = write_dmi &&
(dr_wdata[1:0] == 2'd1 || dr_wdata[1:0] == 2'd2) &&
!(dmi_busy || dmi_cmderr != 2'd0) && dtm_pready;
assign dtm_penable = 1'b0;
// paddr/pwdata/pwrite are valid momentarily when psel is asserted.
assign dtm_paddr = dr_wdata[34 +: W_ADDR];
assign dtm_pwrite = dr_wdata[1];
assign dtm_pwdata = dr_wdata[2 +: 32];
always @ (posedge tck or negedge trst_n) begin
if (!trst_n) begin
dmi_busy <= 1'b0;
dmi_cmderr <= 2'd0;
end else if (read_dmi) begin
// Reading while busy sets the busy sticky error. Note the capture
// into shift register should also reflect this update on-the-fly
if (dmi_busy && dmi_cmderr == 2'd0)
dmi_cmderr <= 2'h3;
end else if (write_dtmcs) begin
// Writing dtmcs.dmireset = 1 clears a sticky error
if (dr_wdata[16])
dmi_cmderr <= 2'd0;
end else if (write_dmi) begin
if (dtm_psel) begin
dmi_busy <= 1'b1;
end else if (dr_wdata[1:0] != 2'd0) begin
// DMI ignored operation, so set sticky busy
if (dmi_cmderr == 2'd0)
dmi_cmderr <= 2'd3;
end
end else if (dmi_busy && dtm_pready) begin
dmi_busy <= 1'b0;
if (dmi_cmderr == 2'd0 && dtm_pslverr)
dmi_cmderr <= 2'd2;
end
end
// DTM logic is in TCK domain, actual DMI + DM is in processor domain
hazard3_apb_async_bridge #(
.W_ADDR (W_ADDR),
.W_DATA (32),
.N_SYNC_STAGES (2)
) inst_hazard3_apb_async_bridge (
.clk_src (tck),
.rst_n_src (trst_n),
.clk_dst (clk_dmi),
.rst_n_dst (rst_n_dmi),
.src_psel (dtm_psel),
.src_penable (dtm_penable),
.src_pwrite (dtm_pwrite),
.src_paddr (dtm_paddr),
.src_pwdata (dtm_pwdata),
.src_prdata (dtm_prdata),
.src_pready (dtm_pready),
.src_pslverr (dtm_pslverr),
.dst_psel (dmi_psel),
.dst_penable (dmi_penable),
.dst_pwrite (dmi_pwrite),
.dst_paddr (dmi_paddr),
.dst_pwdata (dmi_pwdata),
.dst_prdata (dmi_prdata),
.dst_pready (dmi_pready),
.dst_pslverr (dmi_pslverr)
);
// ----------------------------------------------------------------------------
// DR read/write
wire [W_DR_SHIFT-1:0] dtmcs_rdata = {
{W_ADDR{1'b0}},
19'h0,
DTMCS_IDLE_HINT[2:0],
dmi_cmderr,
W_ADDR[5:0], // abits
4'd1 // version
};
wire [W_DR_SHIFT-1:0] dmi_rdata = {
{W_ADDR{1'b0}},
dtm_prdata,
dmi_busy && dmi_cmderr == 2'd0 ? 2'd3 : dmi_cmderr
};
assign dr_rdata = dr_sel_dmi_ndtmcs ? dmi_rdata : dtmcs_rdata;
always @ (posedge tck or negedge trst_n) begin
if (!trst_n) begin
dmihardreset_req <= 1'b0;
end else begin
dmihardreset_req <= write_dtmcs && dr_wdata[17];
end
end
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+162
View File
@@ -0,0 +1,162 @@
/*****************************************************************************\
| Copyright (C) 2021-2025 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Implement a RISC-V JTAG DTM tunnelled through a Xilinx BSCANE2 primitive.
//
// Xilinx allows up to four custom DRs to be added to the FPGA TAP controller.
// A JTAG-DTM only needs two: DTMCS and DMI.
//
// With the correct config, OpenOCD can treat the FPGA TAP as a JTAG DTM and
// access RISC-V debug directly. This allows you to debug internal RISC-V
// cores with the same JTAG interface you use to load the FPGA.
//
// CHAIN_DTMCS and CHAIN_DMI select which JTAG IR values are used to access
// these DRs. Values 1 through 4 correspond to Xilinx USER1 through USER4
// instructions, which have IR values 0x02, 0x03, 0x22, 0x23.
`default_nettype none
module hazard3_xilinx7_jtag_dtm #(
parameter SEL_DTMCS = 3,
parameter SEL_DMI = 4,
parameter DTMCS_IDLE_HINT = 3'd4,
parameter W_PADDR = 9,
parameter ABITS = W_PADDR - 2 // do not modify
) (
// This is synchronous to TCK and asserted for one TCK cycle only
output wire dmihardreset_req,
// Bus clock + reset for Debug Module Interface
input wire clk_dmi,
input wire rst_n_dmi,
// Debug Module Interface (APB)
output wire dmi_psel,
output wire dmi_penable,
output wire dmi_pwrite,
output wire [W_PADDR-1:0] dmi_paddr,
output wire [31:0] dmi_pwdata,
input wire [31:0] dmi_prdata,
input wire dmi_pready,
input wire dmi_pslverr
);
// Signals to/from the Xilinx TAP
wire jtck_unbuf;
wire jtck;
wire jtdo2;
wire jtdo1;
wire jtdi;
wire jshift;
wire jupdate;
wire jcapture;
wire jrst;
wire jrst_n = !jrst;
wire jce2;
wire jce1;
BSCANE2 #(
.JTAG_CHAIN (SEL_DTMCS) // Value for USER command.
) bscan_dtmcs (
.CAPTURE (jcapture), // CAPTURE output from TAP controller.
.DRCK (/* unused */), // Gated TCK output. When SEL is asserted, DRCK toggles when CAPTURE or SHIFT are asserted.
.RESET (jrst), // Reset output for TAP controller.
.RUNTEST (/* unused */), // Output asserted when TAP controller is in Run Test/Idle state.
.SEL (jce1), // USER instruction active output.
.SHIFT (jshift), // SHIFT output from TAP controller.
.TCK (jtck_unbuf), // Test Clock output. Fabric connection to TAP Clock pin.
.TDI (jtdi), // Test Data Input (TDI) output from TAP controller.
.TMS (/* unused */), // Test Mode Select output. Fabric connection to TAP.
.UPDATE (jupdate), // UPDATE output from TAP controller
.TDO (jtdo1) // Test Data Output (TDO) input for USER function.
);
BSCANE2 #(
.JTAG_CHAIN (SEL_DMI)
) bscan_dmi (
.CAPTURE (/* unused */),
.DRCK (/* unused */),
.RESET (/* unused */),
.RUNTEST (/* unused */),
.SEL (jce2),
.SHIFT (/* unused */),
.TCK (/* unused */),
.TDI (/* unused */),
.TMS (/* unused */),
.UPDATE (/* unused */),
.TDO (jtdo2)
);
BUFG bufg_jtck (
.I (jtck_unbuf),
.O (jtck)
);
localparam W_DR_SHIFT = ABITS + 32 + 2;
wire core_dr_wen = jupdate;
wire core_dr_ren = jcapture;
wire core_dr_sel_dmi_ndtmcs = !jce1;
wire dr_shift_en = jshift;
wire [W_DR_SHIFT-1:0] core_dr_wdata;
wire [W_DR_SHIFT-1:0] core_dr_rdata;
reg [W_DR_SHIFT-1:0] dr_shift;
assign core_dr_wdata = dr_shift;
always @ (posedge jtck or negedge jrst_n) begin
if (!jrst_n) begin
dr_shift <= {W_DR_SHIFT{1'b0}};
end else if (core_dr_ren) begin
dr_shift <= core_dr_rdata;
end else if (dr_shift_en) begin
dr_shift <= {jtdi, dr_shift[W_DR_SHIFT-1:1]};
if (!core_dr_sel_dmi_ndtmcs)
dr_shift[31] <= jtdi;
end
end
// We have only a single shifter for the two DRs, so these are tied together:
assign jtdo1 = dr_shift[0];
assign jtdo2 = dr_shift[0];
// The actual DTM is in here:
hazard3_jtag_dtm_core #(
.DTMCS_IDLE_HINT (DTMCS_IDLE_HINT),
.W_ADDR (ABITS)
) inst_hazard3_jtag_dtm_core (
.tck (jtck),
.trst_n (jrst_n),
.clk_dmi (clk_dmi),
.rst_n_dmi (rst_n_dmi),
.dr_wen (core_dr_wen),
.dr_ren (core_dr_ren),
.dr_sel_dmi_ndtmcs (core_dr_sel_dmi_ndtmcs),
.dr_wdata (core_dr_wdata),
.dr_rdata (core_dr_rdata),
.dmihardreset_req (dmihardreset_req),
.dmi_psel (dmi_psel),
.dmi_penable (dmi_penable),
.dmi_pwrite (dmi_pwrite),
.dmi_paddr (dmi_paddr[W_PADDR-1:2]),
.dmi_pwdata (dmi_pwdata),
.dmi_prdata (dmi_prdata),
.dmi_pready (dmi_pready),
.dmi_pslverr (dmi_pslverr)
);
assign dmi_paddr[1:0] = 2'b00;
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+22
View File
@@ -0,0 +1,22 @@
file hazard3_core.v
file hazard3_cpu_1port.v
file hazard3_cpu_2port.v
file arith/hazard3_alu.v
file arith/hazard3_branchcmp.v
file arith/hazard3_mul_fast.v
file arith/hazard3_muldiv_seq.v
file arith/hazard3_onehot_encode.v
file arith/hazard3_onehot_priority.v
file arith/hazard3_onehot_priority_dynamic.v
file arith/hazard3_priority_encode.v
file arith/hazard3_shift_barrel.v
file hazard3_csr.v
file hazard3_decode.v
file hazard3_frontend.v
file hazard3_instr_decompress.v
file hazard3_irq_ctrl.v
file hazard3_pmp.v
file hazard3_power_ctrl.v
file hazard3_regfile_1w2r.v
file hazard3_triggers.v
include .
+259
View File
@@ -0,0 +1,259 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Hazard3 CPU configuration parameters
// To configure Hazard3 you can either edit this file, or set parameters on
// your top-level instantiation, it's up to you. These parameters are all
// plumbed through Hazard3's internal hierarchy to the appropriate places.
// If you add a parameter here, you should add a matching line to
// hazard3_config_inst.vh to propagate the parameter through module
// instantiations.
// ----------------------------------------------------------------------------
// Reset state configuration
// RESET_VECTOR: Address of first instruction executed.
parameter RESET_VECTOR = 32'h00000000,
// MTVEC_INIT: Initial value of trap vector base. Bits clear in MTVEC_WMASK
// will never change from this initial value. Bits set in MTVEC_WMASK can be
// written/set/cleared as normal.
//
// Note that mtvec bits 1:0 do not affect the trap base (as per RISC-V spec).
// Bit 1 is don't care, bit 0 selects the vectoring mode: unvectored if == 0
// (all traps go to mtvec), vectored if == 1 (exceptions go to mtvec, IRQs to
// mtvec + mcause * 4). This means MTVEC_INIT also sets the initial vectoring
// mode.
parameter MTVEC_INIT = 32'h00000000,
// ----------------------------------------------------------------------------
// Standard RISC-V ISA support
// EXTENSION_A: Support for atomic read/modify/write instructions
parameter EXTENSION_A = 1,
// EXTENSION_C: Support for compressed (variable-width) instructions
parameter EXTENSION_C = 1,
// EXTENSION_E: Implement the RV32E base extension rather than RV32I. This
// reduces the number of integer registers from 31 to 15.
parameter EXTENSION_E = 0,
// EXTENSION_M: Support for hardware multiply/divide/modulo instructions
parameter EXTENSION_M = 1,
// EXTENSION_ZBA: Support for Zba address generation instructions
parameter EXTENSION_ZBA = 0,
// EXTENSION_ZBB: Support for Zbb basic bit manipulation instructions
parameter EXTENSION_ZBB = 0,
// EXTENSION_ZBC: Support for Zbc carry-less multiplication instructions
parameter EXTENSION_ZBC = 0,
// EXTENSION_ZBKB: Support for Zbkb basic bit manipulation for cryptography
// Requires: Zbb. (This flag enables instructions in Zbkb which aren't in Zbb.)
parameter EXTENSION_ZBKB = 0,
// EXTENSION_ZBKX: support for Zbkx crossbar permutation instructions
parameter EXTENSION_ZBKX = 0,
// EXTENSION_ZBS: Support for Zbs single-bit manipulation instructions
parameter EXTENSION_ZBS = 0,
// EXTENSION_ZCB: Support for Zcb basic additional compressed instructions
// Requires: EXTENSION_C. (Some Zcb instructions also require Zbb or M.)
// Note Zca is equivalent to C, as we do not support the F extension.
parameter EXTENSION_ZCB = 0,
// EXTENSION_ZCLSD: Support for Zclsd compressed load/store pair instructions
// Requires: EXTENSION_ZILSD, EXTENSION_C.
parameter EXTENSION_ZCLSD = 0,
// EXTENSION_ZCMP: Support for Zcmp push/pop instructions.
// Requires: EXTENSION_C.
parameter EXTENSION_ZCMP = 0,
// EXTENSION_ZIFENCEI: Support for the fence.i instruction
// Optional, since a plain branch/jump will also flush the prefetch queue.
parameter EXTENSION_ZIFENCEI = 0,
// EXTENSION_ZILSD: Support for Zilsd load/store pair instructions
parameter EXTENSION_ZILSD = 0,
// ----------------------------------------------------------------------------
// Custom RISC-V extensions
// EXTENSION_XH3B: Custom bit-extract-multiple instructions for Hazard3
parameter EXTENSION_XH3BEXTM = 0,
// EXTENSION_XH3IRQ: Custom preemptive, prioritised interrupt support. Can be
// disabled if an external interrupt controller (e.g. PLIC) is used. If
// disabled, and NUM_IRQS > 1, the external interrupts are simply OR'd into
// mip.meip.
parameter EXTENSION_XH3IRQ = 0,
// EXTENSION_XH3PMPM: PMPCFGMx CSRs to enforce PMP regions in M-mode without
// locking. Unlike ePMP mseccfg.rlb, locked and unlocked regions can coexist
parameter EXTENSION_XH3PMPM = 0,
// EXTENSION_XH3POWER: Custom power management controls for Hazard3
parameter EXTENSION_XH3POWER = 0,
// ----------------------------------------------------------------------------
// Standard CSR support
// Note the Zicsr extension is implied by any of CSR_M_MANDATORY, CSR_M_TRAP,
// CSR_COUNTER.
// CSR_M_MANDATORY: Bare minimum CSR support e.g. misa. Spec says must = 1 if
// CSRs are present, but I won't tell anyone.
parameter CSR_M_MANDATORY = 1,
// CSR_M_TRAP: Include M-mode trap-handling CSRs, and enable trap support.
parameter CSR_M_TRAP = 1,
// CSR_COUNTER: Include performance counters and Zicntr CSRs
parameter CSR_COUNTER = 0,
// U_MODE: Support the U (user) execution mode. In U mode, the core performs
// unprivileged bus accesses, and software's access to CSRs is restricted.
// Additionally, if the PMP is included, the core may restrict U-mode
// software's access to memory.
// Requires: CSR_M_TRAP.
parameter U_MODE = 0,
// PMP_REGIONS: Number of physical memory protection regions, or 0 for no PMP.
// PMP is more useful if U mode is supported, but this is not a requirement.
parameter PMP_REGIONS = 0,
// PMP_GRAIN: This is the "G" parameter in the privileged spec. Minimum PMP
// region size is 1 << (G + 2) bytes. If G > 0, PMCFG.A can not be set to
// NA4 (will get set to OFF instead). If G > 1, the G - 1 LSBs of pmpaddr are
// read-only-0 when PMPCFG.A is OFF, and read-only-1 when PMPCFG.A is NAPOT.
parameter PMP_GRAIN = 0,
// PMP_MATCH_NAPOT: Enable PMP support for the NAPOT (naturally-aligned
// power-of-two) and NA4 (naturally-aligned four-byte) matching modes. When
// disabled, attempting to select these modes will set the PMP region to OFF.
parameter PMP_MATCH_NAPOT = 1,
// PMP_MATCH_TOR: Enable PMP support for the TOR (top-of-range) matching mode.
// When disabled, attempting to select this mode will set the region to OFF.
parameter PMP_MATCH_TOR = 0,
// PMPADDR_HARDWIRED: If a bit is 1, the corresponding region's pmpaddr and
// pmpcfg registers are read-only. PMP_GRAIN is ignored on hardwired regions.
// It's recommended to make hardwired regions the highest-numbered, so they
// can be overridden by lower-numbered regions.
parameter PMP_HARDWIRED = {(PMP_REGIONS > 0 ? PMP_REGIONS : 1){1'b0}},
// PMPADDR_HARDWIRED_ADDR: Values of pmpaddr registers whose PMP_HARDWIRED
// bits are set to 1. Non-hardwired regions reset to all-zeroes.
parameter PMP_HARDWIRED_ADDR = {(PMP_REGIONS > 0 ? PMP_REGIONS : 1){32'h0}},
// PMPCFG_RESET_VAL: Values of pmpcfg registers whose PMP_HARDWIRED bits are
// set to 1. Non-hardwired regions reset to all zeroes.
parameter PMP_HARDWIRED_CFG = {(PMP_REGIONS > 0 ? PMP_REGIONS : 1){8'h00}},
// DEBUG_SUPPORT: Support for run/halt and instruction injection from an
// external Debug Module, support for Debug Mode, and Debug Mode CSRs.
// Requires: CSR_M_MANDATORY, CSR_M_TRAP.
parameter DEBUG_SUPPORT = 0,
// BREAKPOINT_TRIGGERS: Number of triggers which support type=2 execute=1
// (but not store/load=1, i.e. not a watchpoint). Requires: DEBUG_SUPPORT
parameter BREAKPOINT_TRIGGERS = 0,
// ----------------------------------------------------------------------------
// External interrupt support
// NUM_IRQS: Number of external IRQs. Minimum 1, maximum 512. Note that if
// EXTENSION_XH3IRQ (Hazard3 interrupt controller) is disabled then multiple
// external interrupts are simply OR'd into mip.meip.
parameter NUM_IRQS = 1,
// IRQ_PRIORITY_BITS: Number of priority bits implemented for each interrupt
// in meipra, if EXTENSION_XH3IRQ is enabled. The number of distinct levels
// is (1 << IRQ_PRIORITY_BITS). Minimum 0, max 4. Note that multiple priority
// levels with a large number of IRQs will have a severe effect on timing.
parameter IRQ_PRIORITY_BITS = 0,
// IRQ_INPUT_BYPASS: disable the input registers on the external interrupts,
// to reduce latency by one cycle. Can be applied on an IRQ-by-IRQ basis.
// Ignored if EXTENSION_XH3IRQ is disabled.
parameter IRQ_INPUT_BYPASS = {(NUM_IRQS > 0 ? NUM_IRQS : 1){1'b0}},
// ----------------------------------------------------------------------------
// ID registers
// JEDEC JEP106-compliant vendor ID, can be left at 0 if "not implemented or
// [...] this is a non-commercial implementation" (RISC-V spec).
// 31:7 is continuation code count, 6:0 is ID. Parity bit is not stored.
parameter MVENDORID_VAL = 32'h0,
// Pointer to configuration structure blob, or all-zeroes. Must be at least
// 4-byte-aligned.
parameter MCONFIGPTR_VAL = 32'h0,
// ----------------------------------------------------------------------------
// Performance/size options
// REDUCED_BYPASS: Remove all forwarding paths except X->X (so back-to-back
// ALU ops can still run at 1 CPI), to save area.
parameter REDUCED_BYPASS = 0,
// MULDIV_UNROLL: Bits per clock for multiply/divide circuit, if present. Must
// be a power of 2.
parameter MULDIV_UNROLL = 1,
// MUL_FAST: Use single-cycle multiply circuit for MUL instructions, retiring
// to stage 3. The sequential multiply/divide circuit is still used for MULH*
parameter MUL_FAST = 0,
// MUL_FASTER: Retire fast multiply results to stage 2 instead of stage 3.
// Throughput is the same, but latency is reduced from 2 cycles to 1 cycle.
// Requires: MUL_FAST.
parameter MUL_FASTER = 0,
// MULH_FAST: extend the fast multiply circuit to also cover MULH*, and remove
// the multiply functionality from the sequential multiply/divide circuit.
// Requires: MUL_FAST
parameter MULH_FAST = 0,
// FAST_BRANCHCMP: Instantiate a separate comparator (eq/lt/ltu) for branch
// comparisons, rather than using the ALU. Improves fetch address delay,
// especially if Zba extension is enabled. Disabling may save area.
parameter FAST_BRANCHCMP = 1,
// RESET_REGFILE: whether to support reset of the general purpose registers.
// There are around 1k bits in the register file, so the reset can be
// disabled e.g. to permit block-RAM inference on FPGA.
parameter RESET_REGFILE = 0,
// BRANCH_PREDICTOR: enable branch prediction. The branch predictor consists
// of a single BTB entry which is allocated on a taken backward branch, and
// cleared on a mispredicted nontaken branch, a fence.i or a trap. Successful
// prediction eliminates the 1-cyle fetch bubble on a taken branch, usually
// making tight loops faster.
parameter BRANCH_PREDICTOR = 0,
// MTVEC_WMASK: Mask of which bits in mtvec are writable. Full writability is
// recommended, because a common idiom in setup code is to set mtvec just
// past code that may trap, as a hardware "try {...} catch" block.
//
// - The vectoring mode can be made fixed by clearing the LSB of MTVEC_WMASK
//
// - In vectored mode, the vector table must be aligned to its size, rounded
// up to a power of two.
parameter MTVEC_WMASK = 32'hfffffffd,
// ----------------------------------------------------------------------------
// Port size parameters (do not modify)
parameter W_ADDR = 32, // Do not modify
parameter W_DATA = 32 // Do not modify
+59
View File
@@ -0,0 +1,59 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Pass-through of parameters defined in hazard3_config.vh, so that these can
// be set at instantiation rather than editing the config file, and will flow
// correctly down through the hierarchy.
.RESET_VECTOR (RESET_VECTOR),
.MTVEC_INIT (MTVEC_INIT),
.EXTENSION_A (EXTENSION_A),
.EXTENSION_C (EXTENSION_C),
.EXTENSION_E (EXTENSION_E),
.EXTENSION_M (EXTENSION_M),
.EXTENSION_ZBA (EXTENSION_ZBA),
.EXTENSION_ZBB (EXTENSION_ZBB),
.EXTENSION_ZBC (EXTENSION_ZBC),
.EXTENSION_ZBKB (EXTENSION_ZBKB),
.EXTENSION_ZBKX (EXTENSION_ZBKX),
.EXTENSION_ZBS (EXTENSION_ZBS),
.EXTENSION_ZCB (EXTENSION_ZCB),
.EXTENSION_ZCLSD (EXTENSION_ZCLSD),
.EXTENSION_ZCMP (EXTENSION_ZCMP),
.EXTENSION_ZIFENCEI (EXTENSION_ZIFENCEI),
.EXTENSION_ZILSD (EXTENSION_ZILSD),
.EXTENSION_XH3BEXTM (EXTENSION_XH3BEXTM),
.EXTENSION_XH3IRQ (EXTENSION_XH3IRQ),
.EXTENSION_XH3PMPM (EXTENSION_XH3PMPM),
.EXTENSION_XH3POWER (EXTENSION_XH3POWER),
.CSR_M_MANDATORY (CSR_M_MANDATORY),
.CSR_M_TRAP (CSR_M_TRAP),
.CSR_COUNTER (CSR_COUNTER),
.U_MODE (U_MODE),
.PMP_REGIONS (PMP_REGIONS),
.PMP_GRAIN (PMP_GRAIN),
.PMP_MATCH_NAPOT (PMP_MATCH_NAPOT),
.PMP_MATCH_TOR (PMP_MATCH_TOR),
.PMP_HARDWIRED (PMP_HARDWIRED),
.PMP_HARDWIRED_ADDR (PMP_HARDWIRED_ADDR),
.PMP_HARDWIRED_CFG (PMP_HARDWIRED_CFG),
.DEBUG_SUPPORT (DEBUG_SUPPORT),
.BREAKPOINT_TRIGGERS (BREAKPOINT_TRIGGERS),
.NUM_IRQS (NUM_IRQS),
.IRQ_PRIORITY_BITS (IRQ_PRIORITY_BITS),
.IRQ_INPUT_BYPASS (IRQ_INPUT_BYPASS),
.MVENDORID_VAL (MVENDORID_VAL),
.MCONFIGPTR_VAL (MCONFIGPTR_VAL),
.REDUCED_BYPASS (REDUCED_BYPASS),
.MULDIV_UNROLL (MULDIV_UNROLL),
.MUL_FAST (MUL_FAST),
.MUL_FASTER (MUL_FASTER),
.MULH_FAST (MULH_FAST),
.FAST_BRANCHCMP (FAST_BRANCHCMP),
.BRANCH_PREDICTOR (BRANCH_PREDICTOR),
.MTVEC_WMASK (MTVEC_WMASK),
.RESET_REGFILE (RESET_REGFILE),
.W_ADDR (W_ADDR),
.W_DATA (W_DATA)
+1590
View File
File diff suppressed because it is too large Load Diff
+343
View File
@@ -0,0 +1,343 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Single-ported top level file for Hazard3 CPU. This file instantiates the
// Hazard3 core, and arbitrates its instruction fetch and load/store signals
// down to a single AHB5 master port.
`ifdef HAZARD3_RVFI_STANDALONE
`include "hazard3_rvfi_standalone_defs.vh"
`endif
`default_nettype none
module hazard3_cpu_1port #(
`include "hazard3_config.vh"
) (
// Global signals
input wire clk,
input wire clk_always_on,
input wire rst_n,
`ifdef RISCV_FORMAL
`RVFI_OUTPUTS ,
`endif
// Power control signals
output wire pwrup_req,
input wire pwrup_ack,
output wire clk_en,
output wire unblock_out,
input wire unblock_in,
// AHB5 Master port
output reg [W_ADDR-1:0] haddr,
output reg hwrite,
output reg [1:0] htrans,
output reg [2:0] hsize,
output wire [2:0] hburst,
output reg [3:0] hprot,
output wire hmastlock,
output reg [7:0] hmaster,
output reg hexcl,
input wire hready,
input wire hresp,
input wire hexokay,
output wire [W_DATA-1:0] hwdata,
input wire [W_DATA-1:0] hrdata,
// Memory ordering signals
output wire fence_i_vld,
output wire fence_d_vld,
input wire fence_rdy,
// Debugger run/halt control
input wire dbg_req_halt,
input wire dbg_req_halt_on_reset,
input wire dbg_req_resume,
output wire dbg_halted,
output wire dbg_running,
// Debugger access to data0 CSR
input wire [W_DATA-1:0] dbg_data0_rdata,
output wire [W_DATA-1:0] dbg_data0_wdata,
output wire dbg_data0_wen,
// Debugger instruction injection
input wire [W_DATA-1:0] dbg_instr_data,
input wire dbg_instr_data_vld,
output wire dbg_instr_data_rdy,
output wire dbg_instr_caught_exception,
output wire dbg_instr_caught_ebreak,
// Optional debug system bus access patch-through
input wire [W_ADDR-1:0] dbg_sbus_addr,
input wire dbg_sbus_write,
input wire [1:0] dbg_sbus_size,
input wire dbg_sbus_vld,
output wire dbg_sbus_rdy,
output wire dbg_sbus_err,
input wire [W_DATA-1:0] dbg_sbus_wdata,
output wire [W_DATA-1:0] dbg_sbus_rdata,
// Identification CSR values
input wire [W_DATA-1:0] mhartid_val,
input wire [3:0] eco_version,
// Level-sensitive interrupt sources
input wire [NUM_IRQS-1:0] irq, // -> mip.meip
input wire soft_irq, // -> mip.msip
input wire timer_irq // -> mip.mtip
);
// ----------------------------------------------------------------------------
// Processor core
// Instruction fetch signals
wire core_aph_req_i;
wire core_aph_panic_i;
wire core_aph_ready_i;
wire core_dph_ready_i;
wire core_dph_err_i;
wire [W_ADDR-1:0] core_haddr_i;
wire [2:0] core_hsize_i;
wire core_priv_i;
wire [W_DATA-1:0] core_rdata_i;
// Load/store signals
wire core_aph_req_d;
wire core_aph_excl_d;
wire core_aph_ready_d;
wire core_dph_ready_d;
wire core_dph_err_d;
wire core_dph_exokay_d;
wire [W_ADDR-1:0] core_haddr_d;
wire [2:0] core_hsize_d;
wire core_priv_d;
wire core_hwrite_d;
wire [W_DATA-1:0] core_wdata_d;
wire [W_DATA-1:0] core_rdata_d;
hazard3_core #(
`include "hazard3_config_inst.vh"
) core (
.clk (clk),
.clk_always_on (clk_always_on),
.rst_n (rst_n),
.pwrup_req (pwrup_req),
.pwrup_ack (pwrup_ack),
.clk_en (clk_en),
.unblock_out (unblock_out),
.unblock_in (unblock_in),
`ifdef RISCV_FORMAL
`RVFI_CONN ,
`endif
.bus_aph_req_i (core_aph_req_i),
.bus_aph_panic_i (core_aph_panic_i),
.bus_aph_ready_i (core_aph_ready_i),
.bus_dph_ready_i (core_dph_ready_i),
.bus_dph_err_i (core_dph_err_i),
.bus_haddr_i (core_haddr_i),
.bus_hsize_i (core_hsize_i),
.bus_priv_i (core_priv_i),
.bus_rdata_i (core_rdata_i),
.bus_aph_req_d (core_aph_req_d),
.bus_aph_excl_d (core_aph_excl_d),
.bus_aph_ready_d (core_aph_ready_d),
.bus_dph_ready_d (core_dph_ready_d),
.bus_dph_err_d (core_dph_err_d),
.bus_dph_exokay_d (core_dph_exokay_d),
.bus_haddr_d (core_haddr_d),
.bus_hsize_d (core_hsize_d),
.bus_priv_d (core_priv_d),
.bus_hwrite_d (core_hwrite_d),
.bus_wdata_d (core_wdata_d),
.bus_rdata_d (core_rdata_d),
.fence_i_vld (fence_i_vld),
.fence_d_vld (fence_d_vld),
.fence_rdy (fence_rdy),
.dbg_req_halt (dbg_req_halt),
.dbg_req_halt_on_reset (dbg_req_halt_on_reset),
.dbg_req_resume (dbg_req_resume),
.dbg_halted (dbg_halted),
.dbg_running (dbg_running),
.dbg_data0_rdata (dbg_data0_rdata),
.dbg_data0_wdata (dbg_data0_wdata),
.dbg_data0_wen (dbg_data0_wen),
.dbg_instr_data (dbg_instr_data),
.dbg_instr_data_vld (dbg_instr_data_vld),
.dbg_instr_data_rdy (dbg_instr_data_rdy),
.dbg_instr_caught_exception (dbg_instr_caught_exception),
.dbg_instr_caught_ebreak (dbg_instr_caught_ebreak),
.mhartid_val (mhartid_val),
.eco_version (eco_version),
.irq (irq),
.soft_irq (soft_irq),
.timer_irq (timer_irq)
);
// ----------------------------------------------------------------------------
// Arbitration state machine
wire bus_gnt_i;
wire bus_gnt_d;
wire bus_gnt_s;
reg bus_hold_aph;
reg [2:0] bus_gnt_ids_prev;
// Note use of clk_always_on: SBA may use this arbiter to access the bus
// whilst the core is asleep.
always @ (posedge clk_always_on or negedge rst_n) begin
if (!rst_n) begin
bus_hold_aph <= 1'b0;
bus_gnt_ids_prev <= 3'h0;
end else begin
bus_hold_aph <= htrans[1] && !hready && !hresp;
bus_gnt_ids_prev <= {bus_gnt_i, bus_gnt_d, bus_gnt_s};
end
end
// Debug SBA access is lower priority than load/store, but higher than
// instruction fetch. This isn't ideal, but in a tight loop the core may be
// performing an instruction fetch or load/store on every single cycle, and
// this is a simple way to guarantee eventual success of debugger accesses. A
// more complex way would be to add a "panic timer" to boost a stalled sbus
// access over an instruction fetch.
// Note that, often, the sbus will be disconnected: it doesn't provide any
// increase in debugger bus throughput compared with the program buffer and
// autoexec. It's useful for "minimally intrusive" debug bus access(i.e. less
// intrusive than halting the core and resuming it) e.g. for Segger RTT.
reg bus_active_dph_s;
assign {bus_gnt_i, bus_gnt_d, bus_gnt_s} =
bus_hold_aph ? bus_gnt_ids_prev :
core_aph_panic_i ? 3'b100 :
core_aph_req_d ? 3'b010 :
dbg_sbus_vld && !bus_active_dph_s ? 3'b001 :
core_aph_req_i ? 3'b100 :
3'b000 ;
reg bus_active_dph_i;
reg bus_active_dph_d;
always @ (posedge clk_always_on or negedge rst_n) begin
if (!rst_n) begin
bus_active_dph_i <= 1'b0;
bus_active_dph_d <= 1'b0;
bus_active_dph_s <= 1'b0;
end else if (hready) begin
bus_active_dph_i <= bus_gnt_i;
bus_active_dph_d <= bus_gnt_d;
bus_active_dph_s <= bus_gnt_s;
end
end
// ----------------------------------------------------------------------------
// Address phase request muxing
localparam HTRANS_IDLE = 2'b00;
localparam HTRANS_NSEQ = 2'b10;
wire [3:0] hprot_data = {
2'b00, // Noncacheable/nonbufferable
core_priv_d, // Privileged or Normal as per core state
1'b1 // Data access
};
wire [3:0] hprot_instr = {
2'b00, // Noncacheable/nonbufferable
core_priv_i, // Privileged or Normal as per core state
1'b0 // Instruction access
};
wire [3:0] hprot_sbus = {
2'b00, // Noncacheable/nonbufferable
1'b1, // Always privileged
1'b1 // Data access
};
assign hburst = 3'b000; // HBURST_SINGLE
assign hmastlock = 1'b0;
always @ (*) begin
if (bus_gnt_s) begin
htrans = HTRANS_NSEQ;
hexcl = 1'b0;
haddr = dbg_sbus_addr;
hsize = {1'b0, dbg_sbus_size};
hwrite = dbg_sbus_write;
hprot = hprot_sbus;
hmaster = 8'h01;
end else if (bus_gnt_d) begin
htrans = HTRANS_NSEQ;
hexcl = core_aph_excl_d;
haddr = core_haddr_d;
hsize = core_hsize_d;
hwrite = core_hwrite_d;
hprot = hprot_data;
hmaster = 8'h00;
end else if (bus_gnt_i) begin
htrans = HTRANS_NSEQ;
hexcl = 1'b0;
haddr = core_haddr_i;
hsize = core_hsize_i;
hwrite = 1'b0;
hprot = hprot_instr;
hmaster = 8'h00;
end else begin
htrans = HTRANS_IDLE;
hexcl = 1'b0;
haddr = {W_ADDR{1'b0}};
hsize = 3'h0;
hwrite = 1'b0;
hprot = 4'h0;
hmaster = 8'h00;
end
end
assign hwdata = bus_active_dph_s ? dbg_sbus_wdata : core_wdata_d;
// ----------------------------------------------------------------------------
// Response routing
// Data buses directly connected
assign core_rdata_d = hrdata;
assign core_rdata_i = hrdata;
assign dbg_sbus_rdata = hrdata;
// Handhshake based on grant and bus stall
assign core_aph_ready_i = hready && bus_gnt_i;
assign core_dph_ready_i = bus_active_dph_i && hready;
assign core_dph_err_i = bus_active_dph_i && hresp;
// D-side errors are reported even when not ready, so that the core can make
// use of the two-phase error response to cleanly squash a second load/store
// chasing the faulting one down the pipeline.
assign core_aph_ready_d = hready && bus_gnt_d;
assign core_dph_ready_d = bus_active_dph_d && hready;
assign core_dph_err_d = bus_active_dph_d && hresp;
assign core_dph_exokay_d = bus_active_dph_d && hexokay;
assign dbg_sbus_err = bus_active_dph_s && hresp;
assign dbg_sbus_rdy = bus_active_dph_s && hready;
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+327
View File
@@ -0,0 +1,327 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Dual-ported top level file for Hazard3 CPU. This file instantiates the
// Hazard3 core, and interfaces its instruction fetch and load/store signals
// to a pair of AHB5 master ports.
`ifdef HAZARD3_RVFI_STANDALONE
`include "hazard3_rvfi_standalone_defs.vh"
`endif
`default_nettype none
module hazard3_cpu_2port #(
`include "hazard3_config.vh"
) (
// Global signals
input wire clk,
input wire clk_always_on,
input wire rst_n,
// Power control signals
output wire pwrup_req,
input wire pwrup_ack,
output wire clk_en,
output wire unblock_out,
input wire unblock_in,
`ifdef RISCV_FORMAL
`RVFI_OUTPUTS ,
`endif
// Instruction fetch port
output wire [W_ADDR-1:0] i_haddr,
output wire i_hwrite,
output wire [1:0] i_htrans,
output wire [2:0] i_hsize,
output wire [2:0] i_hburst,
output wire [3:0] i_hprot,
output wire i_hmastlock,
output wire [7:0] i_hmaster,
input wire i_hready,
input wire i_hresp,
output wire [W_DATA-1:0] i_hwdata,
input wire [W_DATA-1:0] i_hrdata,
// Load/store port
output wire [W_ADDR-1:0] d_haddr,
output wire d_hwrite,
output wire [1:0] d_htrans,
output wire [2:0] d_hsize,
output wire [2:0] d_hburst,
output wire [3:0] d_hprot,
output wire d_hmastlock,
output wire [7:0] d_hmaster,
output wire d_hexcl,
input wire d_hready,
input wire d_hresp,
input wire d_hexokay,
output wire [W_DATA-1:0] d_hwdata,
input wire [W_DATA-1:0] d_hrdata,
// Memory ordering signals
output wire fence_i_vld,
output wire fence_d_vld,
input wire fence_rdy,
// Debugger run/halt control
input wire dbg_req_halt,
input wire dbg_req_halt_on_reset,
input wire dbg_req_resume,
output wire dbg_halted,
output wire dbg_running,
// Debugger access to data0 CSR
input wire [W_DATA-1:0] dbg_data0_rdata,
output wire [W_DATA-1:0] dbg_data0_wdata,
output wire dbg_data0_wen,
// Debugger instruction injection
input wire [W_DATA-1:0] dbg_instr_data,
input wire dbg_instr_data_vld,
output wire dbg_instr_data_rdy,
output wire dbg_instr_caught_exception,
output wire dbg_instr_caught_ebreak,
// Optional debug system bus access patch-through
input wire [W_ADDR-1:0] dbg_sbus_addr,
input wire dbg_sbus_write,
input wire [1:0] dbg_sbus_size,
input wire dbg_sbus_vld,
output wire dbg_sbus_rdy,
output wire dbg_sbus_err,
input wire [W_DATA-1:0] dbg_sbus_wdata,
output wire [W_DATA-1:0] dbg_sbus_rdata,
// Identification CSR values
input wire [W_DATA-1:0] mhartid_val,
input wire [3:0] eco_version,
// Level-sensitive interrupt sources
input wire [NUM_IRQS-1:0] irq, // -> mip.meip
input wire soft_irq, // -> mip.msip
input wire timer_irq // -> mip.mtip
);
// ----------------------------------------------------------------------------
// Processor core
// Instruction fetch signals
wire core_aph_req_i;
wire core_aph_ready_i;
wire core_dph_ready_i;
wire core_dph_err_i;
wire [W_ADDR-1:0] core_haddr_i;
wire [2:0] core_hsize_i;
wire core_priv_i;
wire [W_DATA-1:0] core_rdata_i;
// Load/store signals
wire core_aph_req_d;
wire core_aph_excl_d;
wire core_aph_ready_d;
wire core_dph_ready_d;
wire core_dph_err_d;
wire core_dph_exokay_d;
wire [W_ADDR-1:0] core_haddr_d;
wire [2:0] core_hsize_d;
wire core_priv_d;
wire core_hwrite_d;
wire [W_DATA-1:0] core_wdata_d;
wire [W_DATA-1:0] core_rdata_d;
hazard3_core #(
`include "hazard3_config_inst.vh"
) core (
.clk (clk),
.clk_always_on (clk_always_on),
.rst_n (rst_n),
.pwrup_req (pwrup_req),
.pwrup_ack (pwrup_ack),
.clk_en (clk_en),
.unblock_out (unblock_out),
.unblock_in (unblock_in),
`ifdef RISCV_FORMAL
`RVFI_CONN ,
`endif
.bus_aph_req_i (core_aph_req_i),
.bus_aph_panic_i (/* unused for 2port */),
.bus_aph_ready_i (core_aph_ready_i),
.bus_dph_ready_i (core_dph_ready_i),
.bus_dph_err_i (core_dph_err_i),
.bus_haddr_i (core_haddr_i),
.bus_hsize_i (core_hsize_i),
.bus_priv_i (core_priv_i),
.bus_rdata_i (core_rdata_i),
.bus_aph_req_d (core_aph_req_d),
.bus_aph_excl_d (core_aph_excl_d),
.bus_aph_ready_d (core_aph_ready_d),
.bus_dph_ready_d (core_dph_ready_d),
.bus_dph_err_d (core_dph_err_d),
.bus_dph_exokay_d (core_dph_exokay_d),
.bus_haddr_d (core_haddr_d),
.bus_hsize_d (core_hsize_d),
.bus_priv_d (core_priv_d),
.bus_hwrite_d (core_hwrite_d),
.bus_wdata_d (core_wdata_d),
.bus_rdata_d (core_rdata_d),
.fence_i_vld (fence_i_vld),
.fence_d_vld (fence_d_vld),
.fence_rdy (fence_rdy),
.dbg_req_halt (dbg_req_halt),
.dbg_req_halt_on_reset (dbg_req_halt_on_reset),
.dbg_req_resume (dbg_req_resume),
.dbg_halted (dbg_halted),
.dbg_running (dbg_running),
.dbg_data0_rdata (dbg_data0_rdata),
.dbg_data0_wdata (dbg_data0_wdata),
.dbg_data0_wen (dbg_data0_wen),
.dbg_instr_data (dbg_instr_data),
.dbg_instr_data_vld (dbg_instr_data_vld),
.dbg_instr_data_rdy (dbg_instr_data_rdy),
.dbg_instr_caught_exception (dbg_instr_caught_exception),
.dbg_instr_caught_ebreak (dbg_instr_caught_ebreak),
.mhartid_val (mhartid_val),
.eco_version (eco_version),
.irq (irq),
.soft_irq (soft_irq),
.timer_irq (timer_irq)
);
// ----------------------------------------------------------------------------
// Instruction port
localparam HTRANS_IDLE = 2'b00;
localparam HTRANS_NSEQ = 2'b10;
assign i_haddr = core_haddr_i;
assign i_htrans = core_aph_req_i ? HTRANS_NSEQ : HTRANS_IDLE;
assign i_hsize = core_hsize_i;
reg dphase_active_i;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
dphase_active_i <= 1'b0;
end else if (i_hready) begin
dphase_active_i <= core_aph_req_i;
end
end
`ifdef HAZARD3_ASSERTIONS
// Wake->sleep transition must wait for outstanding instruction fetches to
// complete, in particular because the arbiter clock will stop
always @ (posedge clk) if (!rst_n) assert(clk_en || !(core_aph_req_i || dphase_active_i));
`endif
assign core_aph_ready_i = i_hready && core_aph_req_i;
assign core_dph_ready_i = i_hready && dphase_active_i;
assign core_dph_err_i = i_hready && dphase_active_i && i_hresp;
assign core_rdata_i = i_hrdata;
assign i_hwrite = 1'b0;
assign i_hburst = 3'h0;
assign i_hmastlock = 1'b0;
assign i_hmaster = 8'h00;
assign i_hwdata = {W_DATA{1'b0}};
assign i_hprot = {
2'b00, // Noncacheable/nonbufferable
core_priv_i, // Privileged or Normal as per core state
1'b0 // Instruction access
};
// ----------------------------------------------------------------------------
// Load/store port
// The debug module has optional System Bus Access support, which can be muxed
// into the processor's D port here (or connected to a standalone AHB shim).
// This confers absolutely no advantage for debugger bus throughput, but
// allows the debugger to access the bus with minimal disturbance to the
// processor.
wire bus_gnt_d;
wire bus_gnt_s;
reg bus_hold_aph;
reg [1:0] bus_gnt_ds_prev;
reg bus_active_dph_d;
reg bus_active_dph_s;
// clk_always_on is used because SBA may access the bus through this arbiter
// whilst the core is asleep (same is not true for I-side interface)
always @ (posedge clk_always_on or negedge rst_n) begin
if (!rst_n) begin
bus_hold_aph <= 1'b0;
bus_gnt_ds_prev <= 2'h0;
end else begin
bus_hold_aph <= d_htrans[1] && !d_hready && !d_hresp;
bus_gnt_ds_prev <= {bus_gnt_d, bus_gnt_s};
end
end
assign {bus_gnt_d, bus_gnt_s} =
bus_hold_aph ? bus_gnt_ds_prev :
core_aph_req_d ? 2'b10 :
dbg_sbus_vld && !bus_active_dph_s ? 2'b01 :
2'b00 ;
always @ (posedge clk_always_on or negedge rst_n) begin
if (!rst_n) begin
bus_active_dph_d <= 1'b0;
bus_active_dph_s <= 1'b0;
end else if (d_hready) begin
bus_active_dph_d <= bus_gnt_d;
bus_active_dph_s <= bus_gnt_s;
end
end
assign d_htrans = bus_gnt_d || bus_gnt_s ? HTRANS_NSEQ : HTRANS_IDLE;
assign d_haddr = bus_gnt_s ? dbg_sbus_addr : core_haddr_d;
assign d_hwrite = bus_gnt_s ? dbg_sbus_write : core_hwrite_d;
assign d_hsize = bus_gnt_s ? {1'b0, dbg_sbus_size} : core_hsize_d;
assign d_hexcl = bus_gnt_s ? 1'b0 : core_aph_excl_d;
assign d_hprot = {
2'b00, // Noncacheable/nonbufferable
bus_gnt_s || core_priv_d, // Privileged or Normal as per core state
1'b1 // Data access
};
assign d_hwdata = bus_active_dph_s ? dbg_sbus_wdata : core_wdata_d;
// D-side errors are reported even when not ready, so that the core can make
// use of the two-phase error response to cleanly squash a second load/store
// chasing the faulting one down the pipeline.
assign core_aph_ready_d = d_hready && bus_gnt_d;
assign core_dph_ready_d = bus_active_dph_d && d_hready;
assign core_dph_err_d = bus_active_dph_d && d_hresp;
assign core_dph_exokay_d = bus_active_dph_d && d_hexokay;
assign core_rdata_d = d_hrdata;
assign dbg_sbus_err = bus_active_dph_s && d_hresp;
assign dbg_sbus_rdy = bus_active_dph_s && d_hready;
assign dbg_sbus_rdata = d_hrdata;
assign d_hburst = 3'h0;
assign d_hmastlock = 1'b0;
assign d_hmaster = bus_gnt_s ? 8'h01 : 8'h00;
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+1642
View File
File diff suppressed because it is too large Load Diff
+204
View File
@@ -0,0 +1,204 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// List of addresses for CSRs implemented by Hazard3, including custom CSRs.
// ----------------------------------------------------------------------------
// M-mode CSRs
// Machine Information Registers (RO)
localparam MVENDORID = 12'hf11; // Vendor ID.
localparam MARCHID = 12'hf12; // Architecture ID.
localparam MIMPID = 12'hf13; // Implementation ID.
localparam MHARTID = 12'hf14; // Hardware thread ID.
localparam MCONFIGPTR = 12'hf15; // Pointer to configuration data structure.
// Machine Trap Setup (RW)
localparam MSTATUS = 12'h300; // Machine status register.
localparam MSTATUSH = 12'h310; // As of priv-1.12 this must be present even if tied 0.
localparam MISA = 12'h301; // ISA and extensions
localparam MEDELEG = 12'h302; // Machine exception delegation register.
localparam MIDELEG = 12'h303; // Machine interrupt delegation register.
localparam MIE = 12'h304; // Machine interrupt-enable register.
localparam MTVEC = 12'h305; // Machine trap-handler base address.
localparam MCOUNTEREN = 12'h306; // Machine counter enable.
// Machine Trap Handling (RW)
localparam MSCRATCH = 12'h340; // Scratch register for machine trap handlers.
localparam MEPC = 12'h341; // Machine exception program counter.
localparam MCAUSE = 12'h342; // Machine trap cause.
localparam MTVAL = 12'h343; // Machine bad address or instruction.
localparam MIP = 12'h344; // Machine interrupt pending.
// Machine Memory Protection (RW)
localparam PMPCFG0 = 12'h3a0; // Physical memory protection configuration.
localparam PMPCFG1 = 12'h3a1; // Physical memory protection configuration, RV32 only.
localparam PMPCFG2 = 12'h3a2; // Physical memory protection configuration.
localparam PMPCFG3 = 12'h3a3; // Physical memory protection configuration, RV32 only.
localparam PMPADDR0 = 12'h3b0; // Physical memory protection address register.
localparam PMPADDR1 = 12'h3b1; // ...
localparam PMPADDR2 = 12'h3b2;
localparam PMPADDR3 = 12'h3b3;
localparam PMPADDR4 = 12'h3b4;
localparam PMPADDR5 = 12'h3b5;
localparam PMPADDR6 = 12'h3b6;
localparam PMPADDR7 = 12'h3b7;
localparam PMPADDR8 = 12'h3b8;
localparam PMPADDR9 = 12'h3b9;
localparam PMPADDR10 = 12'h3ba;
localparam PMPADDR11 = 12'h3bb;
localparam PMPADDR12 = 12'h3bc;
localparam PMPADDR13 = 12'h3bd;
localparam PMPADDR14 = 12'h3be;
localparam PMPADDR15 = 12'h3bf;
localparam MSECCFG = 12'h747;
localparam MSECCFGH = 12'h757;
// Performance counters (RW)
localparam MCYCLE = 12'hb00; // Raw cycles since start of day
localparam MINSTRET = 12'hb02; // Instruction retire count since start of day
localparam MHPMCOUNTER3 = 12'hb03; // WARL (we tie to 0)
localparam MHPMCOUNTER4 = 12'hb04; // WARL (we tie to 0)
localparam MHPMCOUNTER5 = 12'hb05; // WARL (we tie to 0)
localparam MHPMCOUNTER6 = 12'hb06; // WARL (we tie to 0)
localparam MHPMCOUNTER7 = 12'hb07; // WARL (we tie to 0)
localparam MHPMCOUNTER8 = 12'hb08; // WARL (we tie to 0)
localparam MHPMCOUNTER9 = 12'hb09; // WARL (we tie to 0)
localparam MHPMCOUNTER10 = 12'hb0a; // WARL (we tie to 0)
localparam MHPMCOUNTER11 = 12'hb0b; // WARL (we tie to 0)
localparam MHPMCOUNTER12 = 12'hb0c; // WARL (we tie to 0)
localparam MHPMCOUNTER13 = 12'hb0d; // WARL (we tie to 0)
localparam MHPMCOUNTER14 = 12'hb0e; // WARL (we tie to 0)
localparam MHPMCOUNTER15 = 12'hb0f; // WARL (we tie to 0)
localparam MHPMCOUNTER16 = 12'hb10; // WARL (we tie to 0)
localparam MHPMCOUNTER17 = 12'hb11; // WARL (we tie to 0)
localparam MHPMCOUNTER18 = 12'hb12; // WARL (we tie to 0)
localparam MHPMCOUNTER19 = 12'hb13; // WARL (we tie to 0)
localparam MHPMCOUNTER20 = 12'hb14; // WARL (we tie to 0)
localparam MHPMCOUNTER21 = 12'hb15; // WARL (we tie to 0)
localparam MHPMCOUNTER22 = 12'hb16; // WARL (we tie to 0)
localparam MHPMCOUNTER23 = 12'hb17; // WARL (we tie to 0)
localparam MHPMCOUNTER24 = 12'hb18; // WARL (we tie to 0)
localparam MHPMCOUNTER25 = 12'hb19; // WARL (we tie to 0)
localparam MHPMCOUNTER26 = 12'hb1a; // WARL (we tie to 0)
localparam MHPMCOUNTER27 = 12'hb1b; // WARL (we tie to 0)
localparam MHPMCOUNTER28 = 12'hb1c; // WARL (we tie to 0)
localparam MHPMCOUNTER29 = 12'hb1d; // WARL (we tie to 0)
localparam MHPMCOUNTER30 = 12'hb1e; // WARL (we tie to 0)
localparam MHPMCOUNTER31 = 12'hb1f; // WARL (we tie to 0)
localparam MCYCLEH = 12'hb80; // High halves of each counter
localparam MINSTRETH = 12'hb82;
localparam MHPMCOUNTER3H = 12'hb83;
localparam MHPMCOUNTER4H = 12'hb84;
localparam MHPMCOUNTER5H = 12'hb85;
localparam MHPMCOUNTER6H = 12'hb86;
localparam MHPMCOUNTER7H = 12'hb87;
localparam MHPMCOUNTER8H = 12'hb88;
localparam MHPMCOUNTER9H = 12'hb89;
localparam MHPMCOUNTER10H = 12'hb8a;
localparam MHPMCOUNTER11H = 12'hb8b;
localparam MHPMCOUNTER12H = 12'hb8c;
localparam MHPMCOUNTER13H = 12'hb8d;
localparam MHPMCOUNTER14H = 12'hb8e;
localparam MHPMCOUNTER15H = 12'hb8f;
localparam MHPMCOUNTER16H = 12'hb90;
localparam MHPMCOUNTER17H = 12'hb91;
localparam MHPMCOUNTER18H = 12'hb92;
localparam MHPMCOUNTER19H = 12'hb93;
localparam MHPMCOUNTER20H = 12'hb94;
localparam MHPMCOUNTER21H = 12'hb95;
localparam MHPMCOUNTER22H = 12'hb96;
localparam MHPMCOUNTER23H = 12'hb97;
localparam MHPMCOUNTER24H = 12'hb98;
localparam MHPMCOUNTER25H = 12'hb99;
localparam MHPMCOUNTER26H = 12'hb9a;
localparam MHPMCOUNTER27H = 12'hb9b;
localparam MHPMCOUNTER28H = 12'hb9c;
localparam MHPMCOUNTER29H = 12'hb9d;
localparam MHPMCOUNTER30H = 12'hb9e;
localparam MHPMCOUNTER31H = 12'hb9f;
localparam MCOUNTINHIBIT = 12'h320; // Count inhibit register for mcycle/minstret
localparam MHPMEVENT3 = 12'h323; // WARL (we tie to 0)
localparam MHPMEVENT4 = 12'h324; // WARL (we tie to 0)
localparam MHPMEVENT5 = 12'h325; // WARL (we tie to 0)
localparam MHPMEVENT6 = 12'h326; // WARL (we tie to 0)
localparam MHPMEVENT7 = 12'h327; // WARL (we tie to 0)
localparam MHPMEVENT8 = 12'h328; // WARL (we tie to 0)
localparam MHPMEVENT9 = 12'h329; // WARL (we tie to 0)
localparam MHPMEVENT10 = 12'h32a; // WARL (we tie to 0)
localparam MHPMEVENT11 = 12'h32b; // WARL (we tie to 0)
localparam MHPMEVENT12 = 12'h32c; // WARL (we tie to 0)
localparam MHPMEVENT13 = 12'h32d; // WARL (we tie to 0)
localparam MHPMEVENT14 = 12'h32e; // WARL (we tie to 0)
localparam MHPMEVENT15 = 12'h32f; // WARL (we tie to 0)
localparam MHPMEVENT16 = 12'h330; // WARL (we tie to 0)
localparam MHPMEVENT17 = 12'h331; // WARL (we tie to 0)
localparam MHPMEVENT18 = 12'h332; // WARL (we tie to 0)
localparam MHPMEVENT19 = 12'h333; // WARL (we tie to 0)
localparam MHPMEVENT20 = 12'h334; // WARL (we tie to 0)
localparam MHPMEVENT21 = 12'h335; // WARL (we tie to 0)
localparam MHPMEVENT22 = 12'h336; // WARL (we tie to 0)
localparam MHPMEVENT23 = 12'h337; // WARL (we tie to 0)
localparam MHPMEVENT24 = 12'h338; // WARL (we tie to 0)
localparam MHPMEVENT25 = 12'h339; // WARL (we tie to 0)
localparam MHPMEVENT26 = 12'h33a; // WARL (we tie to 0)
localparam MHPMEVENT27 = 12'h33b; // WARL (we tie to 0)
localparam MHPMEVENT28 = 12'h33c; // WARL (we tie to 0)
localparam MHPMEVENT29 = 12'h33d; // WARL (we tie to 0)
localparam MHPMEVENT30 = 12'h33e; // WARL (we tie to 0)
localparam MHPMEVENT31 = 12'h33f; // WARL (we tie to 0)
// Other standard M-mode CSRs:
localparam MENVCFG = 12'h30a;
localparam MENVCFGH = 12'h31a;
// Custom M-mode CSRs:
localparam PMPCFGM0 = 12'hbd0; // Make PMP regions M-mode without locking
// bd1 // (reserved for >32 regions)
localparam MEIEA = 12'hbe0; // External interrupt pending array
localparam MEIPA = 12'hbe1; // External interrupt enable array
localparam MEIFA = 12'hbe2; // External interrupt force array
localparam MEIPRA = 12'hbe3; // External interrupt priority array
localparam MEINEXT = 12'hbe4; // Next external interrupt
localparam MEICONTEXT = 12'hbe5; // External interrupt context register
localparam MSLEEP = 12'hbf0; // M-mode sleep control register
localparam H3MISA = 12'hbf1; // Hazard3 M-mode ISA identification register
// ----------------------------------------------------------------------------
// U-mode CSRs
// Read-only aliases of M-mode counter CSRs:
localparam CYCLE = 12'hc00;
localparam TIME = 12'hc01;
localparam INSTRET = 12'hc02;
localparam CYCLEH = 12'hc80;
localparam TIMEH = 12'hc81;
localparam INSTRETH = 12'hc82;
// Custom U-mode CSRs
localparam SLEEP = 12'h8f0; // U-mode subset of M-mode sleep control
// ----------------------------------------------------------------------------
// Trigger Module
localparam TSELECT = 12'h7a0;
localparam TDATA1 = 12'h7a1;
localparam TDATA2 = 12'h7a2;
localparam TDATA3 = 12'h7a3;
localparam TINFO = 12'h7a4;
localparam TCONTROL = 12'h7a5;
localparam MCONTEXT = 12'h7a8;
// ----------------------------------------------------------------------------
// D-mode CSRs
localparam DCSR = 12'h7b0;
localparam DPC = 12'h7b1;
localparam DMDATA0 = 12'hbff; // Custom read/write
+594
View File
@@ -0,0 +1,594 @@
/*****************************************************************************\
| Copyright (C) 2021-2023 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
`default_nettype none
module hazard3_decode #(
`include "hazard3_config.vh"
,
`include "hazard3_width_const.vh"
) (
input wire clk,
input wire rst_n,
input wire [31:0] fd_cir,
input wire [1:0] fd_cir_err,
input wire [1:0] fd_cir_predbranch,
input wire [1:0] fd_cir_vld,
input wire fd_cir_is_32bit,
input wire fd_cir_invalid_16bit,
input wire fd_cir_is_uop,
input wire fd_cir_uop_nonfinal,
input wire fd_cir_uop_no_pc_update,
input wire fd_cir_uop_atomic,
output wire [1:0] df_cir_use,
output wire df_cir_flush_behind,
output wire df_uop_stall,
output wire df_uop_clear,
output wire df_lspair_phase_next,
output wire [W_ADDR-1:0] d_pc,
input wire debug_mode,
input wire m_mode,
input wire trap_wfi,
input wire [W_ADDR-1:0] debug_dpc_wdata,
input wire debug_dpc_wen,
output wire [W_ADDR-1:0] debug_dpc_rdata,
output wire d_starved,
input wire x_stall,
input wire f_jump_now,
input wire [W_ADDR-1:0] f_jump_target,
input wire x_jump_not_except,
input wire [W_ADDR-1:0] d_btb_target_addr,
output reg [W_DATA-1:0] d_imm,
output reg [W_REGADDR-1:0] d_rs1,
output reg [W_REGADDR-1:0] d_rs2,
output reg [W_REGADDR-1:0] d_rd,
output reg [2:0] d_funct3_32b,
output reg [6:0] d_funct7_32b,
output reg [W_ALUSRC-1:0] d_alusrc_a,
output reg [W_ALUSRC-1:0] d_alusrc_b,
output reg [W_ALUOP-1:0] d_aluop,
output reg [W_MEMOP-1:0] d_memop,
output reg [W_MULOP-1:0] d_mulop,
output reg d_csr_ren,
output reg d_csr_wen,
output reg [1:0] d_csr_wtype,
output reg d_csr_w_imm,
output reg [W_BCOND-1:0] d_branchcond,
output reg [W_ADDR-1:0] d_addr_offs,
output reg d_addr_is_regoffs,
output reg [W_EXCEPT-1:0] d_except,
output reg d_sleep_wfi,
output reg d_sleep_block,
output reg d_sleep_unblock,
output wire d_no_pc_increment,
output wire d_uninterruptible,
output wire [W_ADDR-1:0] d_lspair_offset,
output reg d_fence_i,
output reg d_fence_d
);
`include "rv_opcodes.vh"
`include "hazard3_ops.vh"
localparam HAVE_CSR = CSR_M_MANDATORY || CSR_M_TRAP || CSR_COUNTER;
// ----------------------------------------------------------------------------
wire [31:0] d_instr = fd_cir | {
30'd0, {2{~|EXTENSION_C}}
};
reg d_invalid_32bit;
wire d_invalid = fd_cir_invalid_16bit || d_invalid_32bit;
assign d_uninterruptible = |EXTENSION_ZCMP && fd_cir_uop_atomic;
`ifdef HAZARD3_ASSERTIONS
always @ (posedge clk) if (rst_n) begin
assert(!(d_invalid && fd_cir_is_uop));
assert(!(d_invalid && fd_cir_uop_atomic));
end
`endif
wire d_lspair_nonfinal;
// Signal to null the mepc offset when taking an exception on this
// instruction (because uops in a sequence *which can except*, so excluding
// the final sp adjust on popret/popretz, will all have the same PC as the
// next uop, which will be in stage 2 when they take their exception)
assign d_no_pc_increment = fd_cir_uop_nonfinal || d_lspair_nonfinal;
assign df_uop_stall = x_stall || d_starved;
// Note !df_cir_flush_behind because the jump in cm.popret/popretz is the
// *penultimate* instruction: we execute the stack adjustment in the fetch
// bubble to save a cycle, still need to finish the uop sequence.
//
// The sp adjust cannot generate an exception (it's an `add` with the same
// PMP.X and breakpoint comparison results as earlier uops) and interrupts are
// suppressed for this part of the sequence.
assign df_uop_clear = f_jump_now && !df_cir_flush_behind;
// Decode various immediate formats
wire [31:0] d_imm_i = {{21{d_instr[31]}}, d_instr[30:20]};
wire [31:0] d_imm_s = {{21{d_instr[31]}}, d_instr[30:25], d_instr[11:7]};
wire [31:0] d_imm_b = {{20{d_instr[31]}}, d_instr[7], d_instr[30:25], d_instr[11:8], 1'b0};
wire [31:0] d_imm_u = {d_instr[31:12], {12{1'b0}}};
wire [31:0] d_imm_j = {{12{d_instr[31]}}, d_instr[19:12], d_instr[20], d_instr[30:21], 1'b0};
// ----------------------------------------------------------------------------
// PC/CIR control
// Must not flag bus error for a valid 16-bit instruction *followed by* an
// error, because instruction fetch errors are speculative, and can be
// flushed by e.g. a branch instruction. Note the 16 LSBs must be valid for
// us to know an instruction's size.
wire d_except_instr_bus_fault = fd_cir_vld > 2'd0 && fd_cir_err[0] ||
fd_cir_vld > 2'd1 && fd_cir_is_32bit && fd_cir_err[1];
assign d_starved = ~|fd_cir_vld || fd_cir_vld[0] && fd_cir_is_32bit;
wire d_stall = x_stall || d_starved || fd_cir_uop_nonfinal || d_lspair_nonfinal;
assign df_cir_use =
d_starved || d_stall ? 2'h0 :
fd_cir_is_32bit ? 2'h2 : 2'h1;
// CIR Locking is required if we successfully assert a jump request, but
// decode is stalled. It is not possible to gate the jump request if the
// stall depends on bus stall (as this would create a through-path from bus
// stall to bus request) so instead we instruct the frontend to preserve the
// stalled instruction when flushing, and fill in behind it.
//
// Once the stall clears, the stalled instruction can execute its remaining
// side effects e.g. writing a link value to the register file.
wire jump_caused_by_d = f_jump_now && x_jump_not_except;
wire assert_cir_lock = jump_caused_by_d && d_stall;
// CIR lock ends naturally when an instruction (not just uop) graduates to the
// next stage:
wire finished_cir_lock = !d_stall;
// CIR lock can meet an untimely end due to trap entry. One way to reach this
// is a dphase load fault on the final load in a cm.popret: here the `ret`
// issues a fetch address while stalled on the first dphase cycle, then is
// flushed by trap on second cycle.
wire deassert_cir_lock = finished_cir_lock || (f_jump_now && !x_jump_not_except);
reg cir_lock_prev;
wire cir_lock = (cir_lock_prev && !deassert_cir_lock) || assert_cir_lock;
assign df_cir_flush_behind = assert_cir_lock && !cir_lock_prev;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
cir_lock_prev <= 1'b0;
end else begin
cir_lock_prev <= cir_lock;
end
end
reg [W_ADDR-1:0] pc;
wire [W_ADDR-1:0] pc_seq_next = pc + (
|EXTENSION_ZCMP && fd_cir_is_uop && fd_cir_uop_no_pc_update ? 32'd0 :
fd_cir_is_32bit ? 32'd4 : 32'd2
);
assign d_pc = pc;
assign debug_dpc_rdata = pc;
// Frontend should mark the whole instruction, and nothing but the
// instruction, as a predicted branch. This goes wrong when we execute the
// address containing the predicted branch twice with different 16-bit
// alignments (!). We need to issue a branch-to-self to get back on a linear
// path, otherwise PC and CIR will diverge and we will misexecute.
wire partial_predicted_branch = !d_starved &&
|BRANCH_PREDICTOR && fd_cir_is_32bit && ^fd_cir_predbranch;
wire predicted_branch = |BRANCH_PREDICTOR && fd_cir_predbranch[0];
// Generally locking takes place on a stalled jump/branch, which may need the
// original PC available to produce a link address when it unstalls. An
// exception to this is jumps in micro-op sequences: in this case the jump is
// the penultimate instruction in the sequence (ret before addi sp) and we
// need to capture the pc mid-uop-sequence.
wire hold_pc_on_cir_lock = assert_cir_lock && !(fd_cir_is_uop && !fd_cir_uop_no_pc_update && !x_stall);
wire update_pc_on_cir_unlock = cir_lock_prev && finished_cir_lock && !fd_cir_uop_no_pc_update;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
pc <= RESET_VECTOR;
end else begin
if (debug_dpc_wen) begin
pc <= debug_dpc_wdata;
end else if (debug_mode) begin
pc <= pc;
end else if ((f_jump_now && !hold_pc_on_cir_lock) || update_pc_on_cir_unlock) begin
pc <= f_jump_target;
end else if (!f_jump_now && fd_cir_uop_nonfinal && !fd_cir_uop_no_pc_update && !x_stall) begin
// End of previously stalled jr uop in cm.popret and cm.popretz:
// safe to update PC as next instruction (addi sp) cannot trap.
pc <= f_jump_target;
end else if (!d_stall && !cir_lock) begin
// If this instruction is a predicted-taken branch (and has not
// generated a mispredict recovery jump) then set PC to the
// prediction target instead of the sequentially next PC
pc <= predicted_branch ? d_btb_target_addr : pc_seq_next;
end
end
end
`ifdef HAZARD3_ASSERTIONS
always @ (posedge clk) if (rst_n) begin
if (~|fd_cir_vld) assert(!fd_cir_is_uop);
if (fd_cir_uop_no_pc_update) assert(fd_cir_is_uop);
if (fd_cir_uop_nonfinal) assert(fd_cir_is_uop);
if ($past(df_uop_clear)) assert(!fd_cir_is_uop);
// Important to avoid spurious PC updates following a trap on the final
// load of a cm.popret:
if ($past(df_uop_clear)) assert(!fd_cir_uop_no_pc_update);
end
`endif
wire [W_ADDR-1:0] branch_offs =
!fd_cir_is_32bit && predicted_branch ? 32'd2 :
fd_cir_is_32bit && predicted_branch ? 32'd4 : d_imm_b;
always @ (*) begin
casez ({|EXTENSION_A, d_instr[6:2]})
{1'bz, 5'b11011}: d_addr_offs = d_imm_j ; // JAL
{1'bz, 5'b11000}: d_addr_offs = branch_offs ; // Branches
{1'bz, 5'b01000}: d_addr_offs = d_imm_s ; // Store
{1'bz, 5'b11001}: d_addr_offs = d_imm_i ; // JALR
{1'bz, 5'b00000}: d_addr_offs = d_imm_i ; // Loads
{1'b1, 5'b01011}: d_addr_offs = 32'h0000_0000; // Atomics
default: d_addr_offs = 32'hxxxx_xxxx;
endcase
if (partial_predicted_branch) begin
d_addr_offs = 32'h0000_0000;
end
end
// ----------------------------------------------------------------------------
// Track phase of load/store pair instructions (Zilsd and Zclsd)
// This could be shared with uop_ctr (for Zcmp) but the two are fundamentally
// different: Zcmp has 16-bit instructions which expand to sequences of
// 32-bit, whereas Zilsd has multi-phase 32-bit instructions and Zclsd has
// direct 16-bit aliases of those instructions. Therefore it's cleaner to
// separate the phasing from the decompression for Zilsd/Zclsd.
wire d_lspair_phase;
// Reorder accesses to avoid clobbering rs1 (base) in first half of load:
wire d_lspair_reg_sel = d_lspair_phase == d_instr[15];
generate
if (EXTENSION_ZILSD) begin: have_lspair_reg_sel
reg d_lspair_phase_r;
assign d_lspair_phase = d_lspair_phase_r;
reg instr_is_lspair;
always @ (*) begin
casez ({d_invalid || d_starved, d_instr})
{1'b0, `RVOPC_LD}: instr_is_lspair = 1'b1;
{1'b0, `RVOPC_SD}: instr_is_lspair = 1'b1;
default: instr_is_lspair = 1'b0;
endcase
end
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
d_lspair_phase_r <= 1'b0;
end else begin
d_lspair_phase_r <= df_lspair_phase_next;
end
end
assign df_lspair_phase_next =
!d_stall || f_jump_now ? 1'b0 :
instr_is_lspair && !x_stall ? 1'b1 : d_lspair_phase_r;
assign d_lspair_nonfinal = instr_is_lspair && !d_lspair_phase_r;
assign d_lspair_offset = {
29'h0,
d_lspair_reg_sel && instr_is_lspair,
2'h0
};
end else begin: no_lspair_reg_sel
assign d_lspair_phase = 1'b0;
assign df_lspair_phase_next = 1'b0;
assign d_lspair_nonfinal = 1'b0;
assign d_lspair_offset = 32'd0;
end
endgenerate
// ----------------------------------------------------------------------------
// Decode X controls
localparam X0 = {W_REGADDR{1'b0}};
// First decode the instruction bits, based on available extensions and
// privilege state, without gating in any stall/exception signals.
reg [W_REGADDR-1:0] raw_rs1;
reg [W_REGADDR-1:0] raw_rs2;
reg [W_REGADDR-1:0] raw_rd;
reg [W_DATA-1:0] raw_imm;
reg [W_ALUSRC-1:0] raw_alusrc_a;
reg [W_ALUSRC-1:0] raw_alusrc_b;
reg [W_ALUOP-1:0] raw_aluop;
reg [W_MEMOP-1:0] raw_memop;
reg [W_MULOP-1:0] raw_mulop;
reg raw_csr_ren;
reg raw_csr_wen;
reg [1:0] raw_csr_wtype;
reg raw_csr_w_imm;
reg [W_BCOND-1:0] raw_branchcond;
reg raw_addr_is_regoffs;
reg [W_EXCEPT-1:0] raw_except;
reg raw_sleep_wfi;
reg raw_sleep_block;
reg raw_sleep_unblock;
reg raw_fence_i;
reg raw_fence_d;
always @ (*) begin
// Assign some defaults
raw_rs1 = d_instr[19:15];
raw_rs2 = d_instr[24:20];
raw_rd = d_instr[11: 7];
raw_imm = d_imm_i;
raw_alusrc_a = ALUSRCA_RS1;
raw_alusrc_b = ALUSRCB_RS2;
raw_aluop = ALUOP_ADD;
raw_memop = MEMOP_NONE;
raw_mulop = M_OP_MUL;
raw_csr_ren = 1'b0;
raw_csr_wen = 1'b0;
raw_csr_wtype = CSR_WTYPE_W;
raw_csr_w_imm = 1'b0;
raw_branchcond = BCOND_NEVER;
raw_addr_is_regoffs = 1'b0;
raw_except = EXCEPT_NONE;
raw_sleep_wfi = 1'b0;
raw_sleep_block = 1'b0;
raw_sleep_unblock = 1'b0;
raw_fence_i = 1'b0;
raw_fence_d = 1'b0;
// Note this funct3/funct7 are valid only for 32-bit instructions. They
// are useful for clusters of related ALU ops, such as sh*add, clmul.
d_funct3_32b = fd_cir[14:12];
d_funct7_32b = fd_cir[31:25];
d_invalid_32bit = 1'b0;
casez (d_instr)
`RVOPC_BEQ: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_SUB; raw_branchcond = BCOND_ZERO; end
`RVOPC_BNE: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_SUB; raw_branchcond = BCOND_NZERO; end
`RVOPC_BLT: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_LT; raw_branchcond = BCOND_NZERO; end
`RVOPC_BGE: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_LT; raw_branchcond = BCOND_ZERO; end
`RVOPC_BLTU: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_LTU; raw_branchcond = BCOND_NZERO; end
`RVOPC_BGEU: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_LTU; raw_branchcond = BCOND_ZERO; end
`RVOPC_JALR: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_branchcond = BCOND_ALWAYS; raw_addr_is_regoffs = 1'b1;
raw_rs2 = X0; raw_aluop = ALUOP_ADD; raw_alusrc_a = ALUSRCA_PC; raw_alusrc_b = ALUSRCB_IMM; raw_imm = fd_cir_is_32bit ? 32'd4 : 32'd2; end
`RVOPC_JAL: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_branchcond = BCOND_ALWAYS; raw_rs1 = X0;
raw_rs2 = X0; raw_aluop = ALUOP_ADD; raw_alusrc_a = ALUSRCA_PC; raw_alusrc_b = ALUSRCB_IMM; raw_imm = fd_cir_is_32bit ? 32'd4 : 32'd2; end
`RVOPC_LUI: begin raw_aluop = ALUOP_RS2; raw_imm = d_imm_u; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; raw_rs1 = X0; end
`RVOPC_AUIPC: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_aluop = ALUOP_ADD; raw_imm = d_imm_u; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; raw_alusrc_a = ALUSRCA_PC; raw_rs1 = X0; end
`RVOPC_ADDI: begin raw_aluop = ALUOP_ADD; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
`RVOPC_SLLI: begin raw_aluop = ALUOP_SLL; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
`RVOPC_SLTI: begin raw_aluop = ALUOP_LT; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
`RVOPC_SLTIU: begin raw_aluop = ALUOP_LTU; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
`RVOPC_XORI: begin raw_aluop = ALUOP_XOR; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
`RVOPC_SRLI: begin raw_aluop = ALUOP_SRL; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
`RVOPC_SRAI: begin raw_aluop = ALUOP_SRA; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
`RVOPC_ORI: begin raw_aluop = ALUOP_OR; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
`RVOPC_ANDI: begin raw_aluop = ALUOP_AND; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
`RVOPC_ADD: begin raw_aluop = ALUOP_ADD; end
`RVOPC_SUB: begin raw_aluop = ALUOP_SUB; end
`RVOPC_SLL: begin raw_aluop = ALUOP_SLL; end
`RVOPC_SLTU: begin raw_aluop = ALUOP_LTU; end
`RVOPC_XOR: begin raw_aluop = ALUOP_XOR; end
`RVOPC_SRL: begin raw_aluop = ALUOP_SRL; end
`RVOPC_SRA: begin raw_aluop = ALUOP_SRA; end
`RVOPC_OR: begin raw_aluop = ALUOP_OR; end
`RVOPC_AND: begin raw_aluop = ALUOP_AND; end
`RVOPC_LB: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LB; end
`RVOPC_LH: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LH; end
`RVOPC_LW: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LW; end
`RVOPC_LBU: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LBU; end
`RVOPC_LHU: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LHU; end
`RVOPC_SB: begin raw_addr_is_regoffs = 1'b1; raw_aluop = ALUOP_RS2; raw_memop = MEMOP_SB; raw_rd = X0; end
`RVOPC_SH: begin raw_addr_is_regoffs = 1'b1; raw_aluop = ALUOP_RS2; raw_memop = MEMOP_SH; raw_rd = X0; end
`RVOPC_SW: begin raw_addr_is_regoffs = 1'b1; raw_aluop = ALUOP_RS2; raw_memop = MEMOP_SW; raw_rd = X0; end
`RVOPC_SLT: begin
raw_aluop = ALUOP_LT;
if (|EXTENSION_XH3POWER && ~|raw_rd && ~|raw_rs1) begin
if (raw_rs2 == 5'h00) begin
// h3.block (power management hint)
d_invalid_32bit = trap_wfi;
raw_sleep_block = !trap_wfi;
end else if (raw_rs2 == 5'h01) begin
// h3.unblock (power management hint)
d_invalid_32bit = trap_wfi;
raw_sleep_unblock = !trap_wfi;
end
end
end
`RVOPC_MUL: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_MUL; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_MULH: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_MULH; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_MULHSU: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_MULHSU; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_MULHU: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_MULHU; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_DIV: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_DIV; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_DIVU: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_DIVU; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_REM: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_REM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_REMU: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_REMU; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_LR_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_LR_W; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_SC_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_SC_W; raw_aluop = ALUOP_RS2; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_AMOSWAP_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_RS2; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_AMOADD_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_ADD; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_AMOXOR_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_XOR; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_AMOAND_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_AND; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_AMOOR_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_OR; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_AMOMIN_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_MIN; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_AMOMAX_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_MAX; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_AMOMINU_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_MINU; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_AMOMAXU_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_MAXU; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_LD: if (EXTENSION_ZILSD) begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LW; raw_rd = {d_instr[11: 8], d_lspair_reg_sel}; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_SD: if (EXTENSION_ZILSD) begin raw_addr_is_regoffs = 1'b1; raw_rd = X0; raw_memop = MEMOP_SW; raw_rs2 = {d_instr[24:21], d_lspair_reg_sel}; raw_aluop = ALUOP_RS2; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_SH1ADD: if (EXTENSION_ZBA) begin raw_aluop = ALUOP_SHXADD; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_SH2ADD: if (EXTENSION_ZBA) begin raw_aluop = ALUOP_SHXADD; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_SH3ADD: if (EXTENSION_ZBA) begin raw_aluop = ALUOP_SHXADD; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_ANDN: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ANDN; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CLZ: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_CLZ; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CPOP: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_CPOP; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CTZ: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_CTZ; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_MAX: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_MAX; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_MAXU: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_MAXU; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_MIN: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_MIN; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_MINU: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_MINU; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_ORC_B: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ORC_B; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_ORN: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ORN; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_REV8: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_REV8; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_ROL: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ROL; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_ROR: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ROR; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_RORI: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ROR; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_SEXT_B: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_SEXT_B; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_SEXT_H: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_SEXT_H; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_XNOR: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_XNOR; end else begin d_invalid_32bit = 1'b1; end
// Note: ZEXT_H is a subset of PACK from Zbkb. This is fine as long
// as this case appears first, since Zbkb implies Zbb on Hazard3.
`RVOPC_ZEXT_H: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ZEXT_H; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CLMUL: if (EXTENSION_ZBC) begin raw_aluop = ALUOP_CLMUL; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CLMULH: if (EXTENSION_ZBC) begin raw_aluop = ALUOP_CLMUL; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CLMULR: if (EXTENSION_ZBC) begin raw_aluop = ALUOP_CLMUL; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_BCLR: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BCLR; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_BCLRI: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BCLR; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_BEXT: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BEXT; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_BEXTI: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BEXT; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_BINV: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BINV; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_BINVI: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BINV; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_BSET: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BSET; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_BSETI: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BSET; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_PACK: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_PACK; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_PACKH: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_PACKH; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_BREV8: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_BREV8; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_UNZIP: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_UNZIP; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_ZIP: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_ZIP; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_XPERM8: if (EXTENSION_ZBKX) begin raw_aluop = ALUOP_XPERM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_XPERM4: if (EXTENSION_ZBKX) begin raw_aluop = ALUOP_XPERM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_H3_BEXTM: if (EXTENSION_XH3BEXTM) begin
raw_aluop = ALUOP_BEXTM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_H3_BEXTMI: if (EXTENSION_XH3BEXTM) begin
raw_aluop = ALUOP_BEXTM; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CSRRW: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = 1'b1 ; raw_csr_ren = |raw_rd; raw_csr_wtype = CSR_WTYPE_W; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CSRRS: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = |raw_rs1; raw_csr_ren = 1'b1 ; raw_csr_wtype = CSR_WTYPE_S; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CSRRC: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = |raw_rs1; raw_csr_ren = 1'b1 ; raw_csr_wtype = CSR_WTYPE_C; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CSRRWI: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = 1'b1 ; raw_csr_ren = |raw_rd; raw_csr_wtype = CSR_WTYPE_W; raw_csr_w_imm = 1'b1; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CSRRSI: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = |raw_rs1; raw_csr_ren = 1'b1 ; raw_csr_wtype = CSR_WTYPE_S; raw_csr_w_imm = 1'b1; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_CSRRCI: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = |raw_rs1; raw_csr_ren = 1'b1 ; raw_csr_wtype = CSR_WTYPE_C; raw_csr_w_imm = 1'b1; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_FENCE: begin raw_rs2 = X0; raw_fence_d = 1'b1; end // Note rs1/rd are zero in instruction
`RVOPC_FENCE_I: if (EXTENSION_ZIFENCEI) begin raw_except = debug_mode ? EXCEPT_NONE : EXCEPT_REFETCH; raw_fence_i = 1'b1; end else begin d_invalid_32bit = 1'b1; end // note rs1/rs2/rd are zero in instruction
`RVOPC_ECALL: if (HAVE_CSR) begin raw_except = m_mode || !U_MODE ? EXCEPT_ECALL_M : EXCEPT_ECALL_U; raw_rs2 = X0; raw_rs1 = X0; raw_rd = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_EBREAK: if (HAVE_CSR) begin raw_except = EXCEPT_EBREAK; raw_rs2 = X0; raw_rs1 = X0; raw_rd = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_MRET: if (HAVE_CSR && m_mode) begin raw_except = EXCEPT_MRET; raw_rs2 = X0; raw_rs1 = X0; raw_rd = X0; end else begin d_invalid_32bit = 1'b1; end
`RVOPC_WFI: if (HAVE_CSR && !trap_wfi) begin raw_sleep_wfi = 1'b1; raw_rs2 = X0; raw_rs1 = X0; raw_rd = X0; end else begin d_invalid_32bit = 1'b1; end
default: begin d_invalid_32bit = 1'b1; end
endcase
if (|EXTENSION_E && (raw_rd[4] || raw_rs1[4] || raw_rs2[4])) begin
d_invalid_32bit = 1'b1;
end
end
// Then gate key signals based on CIR fullness, fetch faults etc. The split
// helps to avoid an event scheduling feedback loop that makes simulators
// unhappy and slow, particularly verilator
localparam [4:0] REGADDR_MASK = {~|EXTENSION_E, 4'hf};
always @ (*) begin
// Pass through by default
d_rs1 = raw_rs1 & REGADDR_MASK;
d_rs2 = raw_rs2 & REGADDR_MASK;
d_rd = raw_rd & REGADDR_MASK;
d_imm = raw_imm;
d_alusrc_a = raw_alusrc_a;
d_alusrc_b = raw_alusrc_b;
d_aluop = raw_aluop;
d_memop = raw_memop;
d_mulop = raw_mulop;
d_csr_ren = raw_csr_ren;
d_csr_wen = raw_csr_wen;
d_csr_wtype = raw_csr_wtype;
d_csr_w_imm = raw_csr_w_imm;
d_branchcond = raw_branchcond;
d_addr_is_regoffs = raw_addr_is_regoffs;
d_except = raw_except;
d_sleep_wfi = raw_sleep_wfi;
d_sleep_block = raw_sleep_block;
d_sleep_unblock = raw_sleep_unblock;
d_fence_i = raw_fence_i;
d_fence_d = raw_fence_d;
if (d_invalid || d_starved || d_except_instr_bus_fault || partial_predicted_branch) begin
d_rs1 = {W_REGADDR{1'b0}};
d_rs2 = {W_REGADDR{1'b0}};
d_rd = {W_REGADDR{1'b0}};
d_memop = MEMOP_NONE;
d_branchcond = BCOND_NEVER;
d_csr_ren = 1'b0;
d_csr_wen = 1'b0;
d_except = EXCEPT_NONE;
d_sleep_wfi = 1'b0;
d_sleep_block = 1'b0;
d_sleep_unblock = 1'b0;
d_fence_i = 1'b0;
d_fence_d = 1'b0;
if (EXTENSION_M)
d_aluop = ALUOP_ADD;
if (d_except_instr_bus_fault)
d_except = EXCEPT_INSTR_FAULT;
else if (d_invalid && !d_starved)
d_except = EXCEPT_INSTR_ILLEGAL;
end
if (partial_predicted_branch) begin
d_addr_is_regoffs = 1'b0;
d_branchcond = BCOND_ALWAYS;
end
if (cir_lock_prev) begin
d_branchcond = BCOND_NEVER;
end
end
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+906
View File
@@ -0,0 +1,906 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
`default_nettype none
module hazard3_frontend #(
`include "hazard3_config.vh"
) (
input wire clk,
input wire rst_n,
// Fetch interface
// addr_vld may be asserted at any time, but after assertion,
// neither addr nor addr_vld may change until the cycle after addr_rdy.
// There is no backpressure on the data interface; the front end
// must ensure it does not request data it cannot receive.
// addr_rdy and dat_vld may be functions of hready, and
// may not be used to compute combinational outputs.
output wire mem_size, // 1'b1 -> 32 bit access
output wire [W_ADDR-1:0] mem_addr,
output wire mem_priv,
output wire mem_addr_vld,
input wire mem_addr_rdy,
input wire [W_DATA-1:0] mem_data,
input wire mem_data_err,
input wire mem_data_vld,
// Jump/flush interface
// Processor may assert vld at any time. The request will not go through
// unless rdy is high. Processor *may* alter request during this time.
// Inputs must not be a function of hready.
input wire [W_ADDR-1:0] jump_target,
input wire jump_priv,
input wire jump_target_vld,
output wire jump_target_rdy,
// Interface to the branch target buffer. `src_addr` is the address of the
// last halfword of a taken backward branch. The frontend redirects fetch
// such that `src_addr` appears to be sequentially followed by `target`.
input wire btb_set,
input wire [W_ADDR-1:0] btb_set_src_addr,
input wire btb_set_src_size,
input wire [W_ADDR-1:0] btb_set_target_addr,
input wire btb_clear,
output wire [W_ADDR-1:0] btb_target_addr_out,
// Interface to Decode
output reg [31:0] cir, // Current instruction register; pre-expanded to 32-bit
output wire [31:0] cir_raw, // Unexpanded instruction data
output reg [1:0] cir_vld, // number of valid halfwords in CIR
input wire [1:0] cir_use, // number of halfwords D intends to consume
// *may* be a function of hready
output wire [1:0] cir_err, // Bus error on upper/lower halfword of CIR.
output wire [1:0] cir_predbranch, // Set for last halfword of a predicted-taken branch
output wire cir_break_any, // Set for exact match of a breakpoint address on CIR LSB
output wire cir_break_d_mode, // As above but specifically break to debug mode
output reg cir_is_32bit, // Can't be decoded from CIR due to pre-expansion
output reg cir_invalid_16bit, // Expanded an invalid 32-bit instruction
output reg cir_is_uop, // Current instruction is part of a micro-op sequence
output reg cir_uop_nonfinal, // ...and there are more to follow in this instruction
output reg cir_uop_no_pc_update, // Suppress PC increment or jump (note the jump in cm.popret is not the final uop!)
output reg cir_uop_atomic, // Prevent IRQ entry, so intermediate states are not observed
input wire uop_stall,
input wire uop_clear,
// "flush_behind": do not flush the oldest instruction when accepting a
// jump request (but still flush younger instructions). Sometimes a
// stalled instruction may assert a jump request, because e.g. the stall
// is dependent on a bus stall signal so can't gate the request.
input wire cir_flush_behind,
// Required for regnum predecode when Zilsd is enabled:
input wire df_lspair_phase_next,
// Signal to power controller that power down is safe. (When going to
// sleep, first the pipeline is stalled, and then the power controller
// waits for the frontend to naturally come to a halt before releasing
// its power request. This avoids manually halting the frontend.)
output wire pwrdown_ok,
// Signal to delay the first instruction fetch following reset, because
// powerup has not yet been negotiated.
input wire delay_first_fetch,
// Provide the rs1/rs2 register numbers which will be in CIR next cycle.
// Coarse: valid if this instruction has a nonzero register operand.
// (Suitable for regfile read)
output reg [4:0] predecode_rs1_coarse,
output reg [4:0] predecode_rs2_coarse,
// Fine: like coarse, but accurate zeroing when the operand is implicit.
// (Suitable for bypass. Still not precise enough for stall logic.)
output reg [4:0] predecode_rs1_fine,
output reg [4:0] predecode_rs2_fine,
// Debugger instruction injection: instruction fetch is suppressed when in
// debug halt state, and the DM can then inject instructions into the last
// entry of the prefetch queue using the vld/rdy handshake.
input wire debug_mode,
input wire [W_DATA-1:0] dbg_instr_data,
input wire dbg_instr_data_vld,
output wire dbg_instr_data_rdy,
// PMP query->kill interface for X permission checks
output wire [W_ADDR-1:0] pmp_i_addr,
output wire pmp_i_m_mode,
input wire pmp_i_kill,
// Trigger unit query->break interface for breakpoints
output wire [W_ADDR-1:0] trigger_addr,
output wire trigger_m_mode,
input wire [1:0] trigger_break_any,
input wire [1:0] trigger_break_d_mode
);
`include "rv_opcodes.vh"
localparam W_BUNDLE = 16;
// This is the minimum for full throughput (enough to avoid dropping data when
// decode stalls) and there is no significant advantage to going larger.
localparam FIFO_DEPTH = 2;
// ----------------------------------------------------------------------------
// Fetch queue
wire jump_now = jump_target_vld && jump_target_rdy;
reg [1:0] mem_data_hwvld;
// PMP X faults are checked in parallel with the fetch (fine if executable
// memory is read-idempotent) and failures are promoted to bus errors:
wire pmp_kill_fetch_dph;
wire mem_or_pmp_err = mem_data_err || pmp_kill_fetch_dph;
// Similarly, breakpoint matches are checked during fetch data phase. These
// are called mem_xxx because they are the breakpoint metadata for the data
// coming back from memory in this dphase.
wire [1:0] mem_break_any;
wire [1:0] mem_break_d_mode;
// Mark data as containing a predicted-taken branch instruction so that
// mispredicts can be recovered -- need to track both halfwords so that we
// can mark the entire instruction, and nothing but the instruction:
reg [1:0] mem_data_predbranch;
// Bus errors (and other metadata) travel alongside data. They cause an
// exception if the core decodes the instruction, but until then can be
// flushed harmlessly.
reg [W_DATA-1:0] fifo_mem [0:FIFO_DEPTH];
reg fifo_err [0:FIFO_DEPTH];
reg [1:0] fifo_break_any [0:FIFO_DEPTH];
reg [1:0] fifo_break_d_mode [0:FIFO_DEPTH];
reg [1:0] fifo_predbranch [0:FIFO_DEPTH];
reg [1:0] fifo_valid_hw [0:FIFO_DEPTH];
reg fifo_valid [0:FIFO_DEPTH];
reg fifo_valid_m1 [0:FIFO_DEPTH];
wire [W_DATA-1:0] fifo_rdata = fifo_mem[0];
wire fifo_full = fifo_valid[FIFO_DEPTH - 1];
wire fifo_empty = !fifo_valid[0];
wire fifo_almost_full = fifo_valid[FIFO_DEPTH - 2];
wire fifo_push;
wire fifo_pop;
wire fifo_dbg_inject = DEBUG_SUPPORT && dbg_instr_data_vld && dbg_instr_data_rdy;
always @ (*) begin: boundary_conditions
integer i;
fifo_mem[FIFO_DEPTH] = mem_data;
fifo_predbranch[FIFO_DEPTH] = 2'b00;
fifo_err[FIFO_DEPTH] = 1'b0;
fifo_break_any[FIFO_DEPTH] = 2'b00;
fifo_break_d_mode[FIFO_DEPTH] = 2'b00;
for (i = 0; i <= FIFO_DEPTH; i = i + 1) begin
fifo_valid[i] = |EXTENSION_C ? |fifo_valid_hw[i] : fifo_valid_hw[i][0];
// valid-to-right condition: i == 0 || fifo_valid[i - 1], but without
// using negative array bound (seems broken in Yosys?) or OOB in the
// short circuit case (gives lint although result is well-defined)
if (i == 0) begin
fifo_valid_m1[i] = 1'b1;
end else begin
fifo_valid_m1[i] = fifo_valid[i - 1];
end
end
end
always @ (posedge clk or negedge rst_n) begin: fifo_update
integer i;
if (!rst_n) begin
for (i = 0; i < FIFO_DEPTH; i = i + 1) begin
fifo_valid_hw[i] <= 2'b00;
fifo_mem[i] <= 32'd0;
fifo_err[i] <= 1'b0;
fifo_break_any[i] <= 2'b00;
fifo_break_d_mode[i] <= 2'b00;
fifo_predbranch[i] <= 2'b00;
end
// This exists only for loop boundary conditions, but is tied off in
// this synchronous process to work around a Verilator scheduling
// issue (see issue #21)
fifo_valid_hw[FIFO_DEPTH] <= 2'b00;
end else begin
for (i = 0; i < FIFO_DEPTH; i = i + 1) begin
if (fifo_pop || (fifo_push && !fifo_valid[i])) begin
fifo_mem[i] <= fifo_valid[i + 1] ? fifo_mem[i + 1] : mem_data;
fifo_err[i] <= fifo_valid[i + 1] ? fifo_err[i + 1] : mem_or_pmp_err;
fifo_break_any[i] <= fifo_valid[i + 1] ? fifo_break_any[i + 1] : mem_break_any;
fifo_break_d_mode[i] <= fifo_valid[i + 1] ? fifo_break_d_mode[i + 1] : mem_break_d_mode;
fifo_predbranch[i] <= fifo_valid[i + 1] ? fifo_predbranch[i + 1] : mem_data_predbranch;
end
fifo_valid_hw[i] <=
jump_now ? 2'h0 :
fifo_valid[i + 1] && fifo_pop ? fifo_valid_hw[i + 1] :
fifo_valid[i] && fifo_pop ? mem_data_hwvld & {2{fifo_push}} :
fifo_valid[i] ? fifo_valid_hw[i] :
fifo_push && !fifo_pop && fifo_valid_m1[i] ? mem_data_hwvld : 2'h0;
end
// Allow DM to inject instructions directly into the lowest-numbered
// queue entry. This mux should not extend critical path since it is
// balanced with the instruction-assembly muxes on the queue bypass
// path. Note that flush takes precedence over debug injection
// (and the debug module design must account for this)
if (fifo_dbg_inject) begin
fifo_mem[0] <= dbg_instr_data;
fifo_err[0] <= 1'b0;
fifo_predbranch[0] <= 2'b00;
fifo_break_any[0] <= 2'b00;
fifo_break_d_mode[0] <= 2'b00;
fifo_valid_hw[0] <= jump_now ? 2'b00 : 2'b11;
end
fifo_valid_hw[FIFO_DEPTH] <= 2'b00;
end
end
`ifdef HAZARD3_ASSERTIONS
always @ (posedge clk) if (rst_n) begin
// FIFO validity must be compact, so we can always consume from the end
if (!fifo_valid[0]) begin
assert(!fifo_valid[1]);
end
end
`endif
assign pwrdown_ok = (fifo_full && !jump_target_vld) || debug_mode;
// ----------------------------------------------------------------------------
// Branch target buffer
wire [W_ADDR-1:0] btb_src_addr;
wire btb_src_size;
wire [W_ADDR-1:0] btb_target_addr;
wire btb_valid;
generate
if (BRANCH_PREDICTOR) begin: have_btb
reg [W_ADDR-1:0] btb_src_addr_r;
reg btb_src_size_r;
reg [W_ADDR-1:0] btb_target_addr_r;
reg btb_valid_r;
assign btb_src_addr = btb_src_addr_r;
assign btb_src_size = btb_src_size_r;
assign btb_target_addr = btb_target_addr_r;
assign btb_valid = btb_valid_r;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
btb_src_addr_r <= {W_ADDR{1'b0}};
btb_src_size_r <= 1'b0;
btb_target_addr_r <= {W_ADDR{1'b0}};
btb_valid_r <= 1'b0;
end else if (btb_clear) begin
// Clear takes precedences over set. E.g. if a taken branch is in
// stage 2 and an exception is in stage 3, we must clear the BTB.
btb_valid_r <= 1'b0;
end else if (btb_set) begin
btb_src_addr_r <= btb_set_src_addr;
btb_src_size_r <= btb_set_src_size;
btb_target_addr_r <= btb_set_target_addr;
btb_valid_r <= 1'b1;
end
end
end else begin: no_btb
assign btb_src_addr = {W_ADDR{1'b0}};
assign btb_src_size = 1'b0;
assign btb_target_addr = {W_ADDR{1'b0}};
assign btb_valid = 1'b0;
end
endgenerate
// Decode uses the target address to set the PC to the correct branch target
// value following a predicted-taken branch (as normally it would update PC
// by following an X jump request, and in this case there is none).
//
// Note this assumes the BTB target has not changed by the time the predicted
// branch arrives at decode! This is always true because the only way for the
// target address to change is when an older branch is taken, which would
// flush the younger predicted-taken branch before it reaches decode.
assign btb_target_addr_out = btb_target_addr;
// ----------------------------------------------------------------------------
// Fetch request generation
// Fetch addr runs ahead of the PC, in word increments.
reg [W_ADDR-1:0] fetch_addr;
reg fetch_priv;
reg btb_prev_start_of_overhanging;
reg [1:0] mem_aph_hwvld;
reg mem_addr_hold;
wire btb_match_word = |BRANCH_PREDICTOR && btb_valid && (
fetch_addr[W_ADDR-1:2] == btb_src_addr[W_ADDR-1:2]
);
// Catch case where predicted-taken branch instruction extends into next word:
wire btb_src_overhanging = btb_src_size && btb_src_addr[1];
// Suppress case where we have jumped immediately after a word-aligned halfword-sized
// branch, and the jump target went into fetch_addr due to an address-phase hold:
wire btb_jumped_beyond = !btb_src_size && !btb_src_addr[1] && !mem_aph_hwvld[0];
wire btb_match_current_addr = btb_match_word && !btb_src_overhanging && !btb_jumped_beyond;
wire btb_match_next_addr = btb_match_word && btb_src_overhanging;
wire btb_match_now = btb_match_current_addr || btb_prev_start_of_overhanging;
// Post-increment if jump request is going straight through
wire [W_ADDR-1:0] jump_target_post_increment =
{jump_target[W_ADDR-1:2], 2'b00} +
{{W_ADDR-3{1'b0}}, mem_addr_rdy && !mem_addr_hold, 2'b00};
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
fetch_addr <= RESET_VECTOR;
// M-mode at reset:
fetch_priv <= 1'b1;
btb_prev_start_of_overhanging <= 1'b0;
end else begin
if (jump_now) begin
fetch_addr <= jump_target_post_increment;
fetch_priv <= jump_priv || !U_MODE;
btb_prev_start_of_overhanging <= 1'b0;
end else if (mem_addr_vld && mem_addr_rdy) begin
if (btb_match_now && |BRANCH_PREDICTOR) begin
fetch_addr <= {btb_target_addr[W_ADDR-1:2], 2'b00};
end else begin
fetch_addr <= fetch_addr + 32'd4;
end
btb_prev_start_of_overhanging <= btb_match_next_addr;
end
end
end
// Combinatorially generate the address-phase request
reg reset_holdoff;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
reset_holdoff <= 1'b1;
end else begin
reset_holdoff <= (|EXTENSION_XH3POWER && delay_first_fetch) ? reset_holdoff : 1'b0;
// This should be impossible, but assert to be sure, because it *will*
// change the fetch address (and we shouldn't check it in hardware if
// we can prove it doesn't happen)
end
end
`ifdef HAZARD3_ASSERTIONS
always @ (posedge clk) if (rst_n) begin
assert(!(jump_target_vld && reset_holdoff));
end
`endif
reg [W_ADDR-1:0] mem_addr_r;
reg mem_priv_r;
reg mem_addr_vld_r;
// Downstream accesses are always word-sized word-aligned.
assign mem_addr = mem_addr_r;
assign mem_priv = mem_priv_r;
assign mem_addr_vld = mem_addr_vld_r && !reset_holdoff;
assign mem_size = 1'b1;
wire fetch_stall;
always @ (*) begin
mem_addr_r = fetch_addr;
mem_priv_r = fetch_priv;
mem_addr_vld_r = 1'b1;
case (1'b1)
mem_addr_hold : begin mem_addr_r = fetch_addr; end
jump_target_vld || reset_holdoff : begin
mem_addr_r = {jump_target[W_ADDR-1:2], 2'b00};
mem_priv_r = jump_priv || !U_MODE;
end
DEBUG_SUPPORT && debug_mode : begin mem_addr_vld_r = 1'b0; end
!fetch_stall : begin mem_addr_r = fetch_addr; end
default : begin mem_addr_vld_r = 1'b0; end
endcase
end
assign jump_target_rdy = !mem_addr_hold;
// ----------------------------------------------------------------------------
// Bus Pipeline Tracking
// Keep track of some useful state of the memory interface
reg [1:0] pending_fetches;
reg [1:0] ctr_flush_pending;
wire [1:0] pending_fetches_next = pending_fetches + (mem_addr_vld && !mem_addr_hold) - mem_data_vld;
// Using the non-registered version of pending_fetches would improve FIFO
// utilisation, but create a combinatorial path from hready to address phase!
// This means at least a 2-word FIFO is required for full fetch throughput.
assign fetch_stall = fifo_full
|| fifo_almost_full && |pending_fetches
|| pending_fetches > 2'h1;
// Debugger only injects instructions when the frontend is at rest and empty.
assign dbg_instr_data_rdy = DEBUG_SUPPORT && !fifo_valid[0] && ~|ctr_flush_pending;
wire cir_room_for_fetch;
// If fetch data is forwarded past the FIFO, ensure it is not also written to it.
assign fifo_push = mem_data_vld && ~|ctr_flush_pending && !(cir_room_for_fetch && fifo_empty)
&& !(DEBUG_SUPPORT && debug_mode);
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
mem_addr_hold <= 1'b0;
pending_fetches <= 2'h0;
ctr_flush_pending <= 2'h0;
end else begin
mem_addr_hold <= mem_addr_vld && !mem_addr_rdy;
pending_fetches <= pending_fetches_next;
if (jump_now) begin
ctr_flush_pending <= pending_fetches - mem_data_vld;
end else if (|ctr_flush_pending && mem_data_vld) begin
ctr_flush_pending <= ctr_flush_pending - 1'b1;
end
end
end
`ifdef HAZARD3_ASSERTIONS
always @ (posedge clk) if (rst_n) begin
assert(ctr_flush_pending <= pending_fetches);
assert(pending_fetches < 2'd3);
assert(!(mem_data_vld && !pending_fetches));
end
`endif
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
mem_data_hwvld <= 2'b11;
mem_aph_hwvld <= 2'b11;
mem_data_predbranch <= 2'b00;
end else begin
if (jump_now) begin
if (|EXTENSION_C) begin
if (mem_addr_rdy) begin
mem_aph_hwvld <= 2'b11;
mem_data_hwvld <= {1'b1, !jump_target[1]};
end else begin
mem_aph_hwvld <= {1'b1, !jump_target[1]};
end
end
mem_data_predbranch <= 2'b00;
end else if (mem_addr_vld && mem_addr_rdy) begin
if (|EXTENSION_C) begin
// If a predicted-taken branch instruction only spans the first
// half of a word, need to flag the second half as invalid.
mem_data_hwvld <= mem_aph_hwvld & {
!(|BRANCH_PREDICTOR && btb_match_now && (btb_src_addr[1] == btb_src_size)),
1'b1
};
// Also need to take the alignment of the destination into account.
mem_aph_hwvld <= {
1'b1,
!(|BRANCH_PREDICTOR && btb_match_now && btb_target_addr[1])
};
end
mem_data_predbranch <=
|BRANCH_PREDICTOR && btb_match_word ? (
btb_src_addr[1] ? 2'b10 :
btb_src_size ? 2'b11 : 2'b01
) :
|BRANCH_PREDICTOR && btb_prev_start_of_overhanging ? (
2'b01
) : 2'b00;
end
end
end
// ----------------------------------------------------------------------------
// PMP and trigger unit interfacing: query -> kill/break
wire [W_ADDR-1:0] pmp_trigger_check_dph_addr;
wire pmp_trigger_check_dph_m_mode;
// Register the fetch address into stage F so that the PMP can check it in
// parallel with the bus data phase. Feels wasteful to have a separate
// register, but using the fetch_addr counter is fraught due to the way that
// new addresses go into it or past it (depending on aphase hold).
generate
if (PMP_REGIONS > 0 || DEBUG_SUPPORT != 0) begin: have_check_reg
reg [W_ADDR-1:0] check_addr_dph;
reg check_m_mode_dph;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
check_addr_dph <= {W_ADDR{1'b0}};
check_m_mode_dph <= 1'b0;
end else if (mem_addr_vld && mem_addr_rdy) begin
check_addr_dph <= mem_addr;
check_m_mode_dph <= mem_priv;
end
end
assign pmp_trigger_check_dph_addr = check_addr_dph;
assign pmp_trigger_check_dph_m_mode = check_m_mode_dph;
end else begin: no_check_reg
assign pmp_trigger_check_dph_addr = {W_ADDR{1'b0}};
assign pmp_trigger_check_dph_m_mode = 1'b0;
end
endgenerate
generate
if (PMP_REGIONS == 0) begin: no_pmp
assign pmp_i_addr = {W_ADDR{1'b0}};
assign pmp_i_m_mode = 1'b0;
assign pmp_kill_fetch_dph = 1'b0;
end else begin: have_pmp
assign pmp_i_addr = pmp_trigger_check_dph_addr;
assign pmp_i_m_mode = pmp_trigger_check_dph_m_mode;
assign pmp_kill_fetch_dph = pmp_i_kill && !debug_mode;
end
endgenerate
generate
if (DEBUG_SUPPORT == 0) begin: no_triggers
assign trigger_addr = {W_ADDR{1'b0}};
assign trigger_m_mode = 1'b0;
assign mem_break_any = 2'b00;
assign mem_break_d_mode = 2'b00;
end else begin: have_triggers
assign trigger_addr = pmp_trigger_check_dph_addr;
assign trigger_m_mode = pmp_trigger_check_dph_m_mode;
assign mem_break_any = trigger_break_any & {|EXTENSION_C, 1'b1};
assign mem_break_d_mode = trigger_break_d_mode & {|EXTENSION_C, 1'b1};
end
endgenerate
// ----------------------------------------------------------------------------
// Instruction buffer
// The instruction buffer is a 3 x ~16-bit shift register:
//
// * 2 x 16-bit entries form the 32-bit current instruction register (CIR)
// which is the processor's decode window
//
// * 1 x 16-bit entry allows the decode window to be non-32-bit-aligned with
// respect to the 2 x 32-bit prefetch queue entries, which are always
// naturally aligned in memory (if fully populated).
//
// The third entry should be trimmed for non-RVC configurations due to
// constant-folding on EXTENSION_C; it is unnecessary here because the
// instructions are always 32-bit-aligned.
// The entries ("slots") are slightly larger than 16 bits because they also
// contain metadata like bus errors:
localparam W_SLOT = 4 + W_BUNDLE;
localparam SLOT_BREAK_ANY_BIT = 3 + W_BUNDLE;
localparam SLOT_BREAK_D_MODE_BIT = 2 + W_BUNDLE;
localparam SLOT_ERR_BIT = 1 + W_BUNDLE;
localparam SLOT_PREDBRANCH_BIT = 0 + W_BUNDLE;
reg [3*W_SLOT-1:0] buf_contents;
reg [1:0] buf_level;
wire fetch_data_vld = !fifo_empty || (mem_data_vld && ~|ctr_flush_pending && !debug_mode);
wire [W_DATA-1:0] fetch_data = fifo_empty ? mem_data : fifo_rdata;
wire [1:0] fetch_data_hwvld = fifo_empty ? mem_data_hwvld : fifo_valid_hw[0];
wire fetch_bus_err = fifo_empty ? mem_or_pmp_err : fifo_err[0];
wire [1:0] fetch_break_any = fifo_empty ? mem_break_any : fifo_break_any[0];
wire [1:0] fetch_break_d_mode = fifo_empty ? mem_break_d_mode : fifo_break_d_mode[0];
wire [1:0] fetch_predbranch = fifo_empty ? mem_data_predbranch : fifo_predbranch[0];
wire [W_SLOT-1:0] fetch_contents_hw1 = {
fetch_break_any[1],
fetch_break_d_mode[1],
fetch_bus_err,
fetch_predbranch[1],
fetch_data[W_BUNDLE +: W_BUNDLE]
};
wire [W_SLOT-1:0] fetch_contents_hw0 = {
fetch_break_any[0],
fetch_break_d_mode[0],
fetch_bus_err,
fetch_predbranch[0],
fetch_data[0 +: W_BUNDLE]
};
wire [2*W_SLOT-1:0] fetch_contents_aligned = {
fetch_contents_hw1,
fetch_data_hwvld[0] || ~|EXTENSION_C ? fetch_contents_hw0 : fetch_contents_hw1
};
// Shift not-yet-used contents down to backfill D's consumption. We don't care
// about anything which is invalid or will be overlaid with fresh data, so
// choose these values in a way that minimises muxes.
wire [3*W_SLOT-1:0] buf_shifted =
cir_use[1] ? {buf_contents[W_SLOT +: 2 * W_SLOT], buf_contents[2 * W_SLOT +: W_SLOT]} :
cir_use[0] && EXTENSION_C ? {buf_contents[2 * W_SLOT +: W_SLOT], buf_contents[W_SLOT +: 2 * W_SLOT]} :
buf_contents;
wire [1:0] level_next_no_fetch = buf_level - cir_use;
// Overlay fresh fetch data onto the shifted/recycled buffer contents. Again,
// if something won't be looked at, generate the cheapest possible garbage.
assign cir_room_for_fetch = level_next_no_fetch <= (|EXTENSION_C && ~&fetch_data_hwvld ? 2'h2 : 2'h1);
assign fifo_pop = cir_room_for_fetch && !fifo_empty;
wire [3*W_SLOT-1:0] buf_shifted_plus_fetch =
!cir_room_for_fetch ? buf_shifted :
level_next_no_fetch[1] && |EXTENSION_C ? {fetch_contents_aligned[0 +: W_SLOT], buf_shifted[0 +: 2 * W_SLOT]} :
level_next_no_fetch[0] && |EXTENSION_C ? {fetch_contents_aligned, buf_shifted[0 +: W_SLOT]} :
{buf_shifted[2 * W_SLOT +: W_SLOT], fetch_contents_aligned};
wire [1:0] fetch_fill_amount = cir_room_for_fetch && fetch_data_vld ? (
&fetch_data_hwvld || ~|EXTENSION_C ? 2'h2 : 2'h1
) : 2'h0;
wire [1:0] buf_level_next = {1'b1, |EXTENSION_C} & (
jump_now && cir_flush_behind ? (cir_is_32bit ? 2'h2 : 2'h1) :
jump_now ? 2'h0 : level_next_no_fetch + fetch_fill_amount
);
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
buf_level <= 2'h0;
cir_vld <= 2'h0;
// Mysterious reset value ensures address buses are zero in reset
// (see definition of d_addr_offs in hazard3_decode)
buf_contents <= {{3 * W_SLOT - 2{1'b0}}, 2'b11};
end else begin
buf_level <= buf_level_next;
cir_vld <= buf_level_next & ~(buf_level_next >> 1'b1);
buf_contents <= buf_shifted_plus_fetch;
end
end
`ifdef HAZARD3_ASSERTIONS
reg [1:0] prop_past_buf_level; // Workaround for weird non-constant $past reset issue
always @ (posedge clk) begin
if (!rst_n) begin
prop_past_buf_level <= 2'h0;
end else begin
prop_past_buf_level <= buf_level;
assert(cir_vld <= 2);
assert(cir_use <= cir_vld);
if (!jump_now) assert(buf_level_next >= level_next_no_fetch);
// We fetch 32 bits per cycle, max. If this happens it's due to negative overflow.
if (prop_past_buf_level == 2'h0)
assert(buf_level != 2'h3);
end
end
`endif
assign cir_err = {
buf_contents[1 * W_SLOT + SLOT_ERR_BIT],
buf_contents[0 * W_SLOT + SLOT_ERR_BIT]
};
assign cir_predbranch = {
buf_contents[1 * W_SLOT + SLOT_PREDBRANCH_BIT],
buf_contents[0 * W_SLOT + SLOT_PREDBRANCH_BIT]
};
assign cir_break_any = buf_contents[0 * W_SLOT + SLOT_BREAK_ANY_BIT] && |cir_vld;
assign cir_break_d_mode = buf_contents[0 * W_SLOT + SLOT_BREAK_D_MODE_BIT] && |cir_vld;
// ----------------------------------------------------------------------------
// Register number predecode
wire [31:0] next_instr = {
buf_shifted_plus_fetch[1 * W_SLOT +: W_BUNDLE],
buf_shifted_plus_fetch[0 * W_SLOT +: W_BUNDLE]
};
wire next_instr_is_32bit = next_instr[1:0] == 2'b11 || ~|EXTENSION_C;
wire [3:0] decomp_uop_step;
wire [3:0] uop_ctr = decomp_uop_step & {4{|EXTENSION_ZCMP}};
wire [4:0] zcmp_pushpop_rs2 =
uop_ctr == 4'h0 ? 5'd01 : // ra
uop_ctr == 4'h1 ? 5'd08 : // s0
uop_ctr == 4'h2 ? 5'd09 : // s1
5'd15 + {1'b0, uop_ctr} ; // s2-s11
wire [4:0] zcmp_pushpop_rs1 =
uop_ctr < 4'hd ? 5'd02 : // sp (addr base reg)
uop_ctr == 4'hd ? 5'd00 : // zero (clear a0)
uop_ctr == 4'he ? 5'd01 : // ra (ret)
5'd02 ; // sp (stack adj)
wire [4:0] zcmp_sa01_r1s = {|next_instr[9:8], ~|next_instr[9:8], next_instr[9:7]};
wire [4:0] zcmp_sa01_r2s = {|next_instr[4:3], ~|next_instr[4:3], next_instr[4:2]};
wire [4:0] zcmp_mvsa01_rs1 = {4'h5, uop_ctr[0]};
wire [4:0] zcmp_mva01s_rs1 = uop_ctr[0] ? zcmp_sa01_r2s : zcmp_sa01_r1s;
// "coarse" because the mapping of pair (x0, x1) -> (x0, x0) is not yet applied
wire [4:0] zilsd_rs2_coarse = { next_instr[24:21], df_lspair_phase_next ^ ~next_instr[15]};
wire [4:0] zclsd_sd_rs2_coarse = {2'b01, next_instr[4:3], df_lspair_phase_next ^ ~next_instr[7] };
wire [4:0] zclsd_sdsp_rs2_coarse = { next_instr[6:3], df_lspair_phase_next ^ 1'b1 };
always @ (*) begin
casez ({next_instr_is_32bit, |EXTENSION_ZCMP, next_instr[15:0]})
{1'b1, 1'bz, 16'bzzzzzzzzzzzzzzzz}: predecode_rs1_coarse = next_instr[19:15]; // 32-bit R, S, B formats
{1'b0, 1'bz, 16'b00zzzzzzzzzzzz00}: predecode_rs1_coarse = 5'd2; // c.addi4spn + don't care
{1'b0, 1'bz, 16'b0zzzzzzzzzzzzz01}: predecode_rs1_coarse = next_instr[11:7]; // c.addi, c.addi16sp + don't care (jal, li)
{1'b0, 1'bz, 16'bz1zzzzzzzzzzzz10}: predecode_rs1_coarse = 5'd2; // c.lwsp, c.swsp, c.ldsp, c.sdsp
{1'b0, 1'bz, 16'bz00zzzzzzzzzzz10}: predecode_rs1_coarse = next_instr[11:7]; // c.slli, c.mv, c.add
{1'b0, 1'b1, 16'b1011zzzzzzzzzz10}: predecode_rs1_coarse = zcmp_pushpop_rs1; // cm.push, cm.pop*
{1'b0, 1'b1, 16'b1010zzzzz0zzzz10}: predecode_rs1_coarse = zcmp_mvsa01_rs1; // cm.mvsa01
{1'b0, 1'b1, 16'b1010zzzzz1zzzz10}: predecode_rs1_coarse = zcmp_mva01s_rs1; // cm.mva01s
default: predecode_rs1_coarse = {2'b01, next_instr[9:7]};
endcase
casez ({next_instr_is_32bit, |EXTENSION_ZCMP, |EXTENSION_ZILSD, |EXTENSION_ZCLSD, next_instr[15:0]})
{1'b1, 1'bz, 1'b1, 1'bz, 16'bzz11zzzzz0z0zzzz}: predecode_rs2_coarse = zilsd_rs2_coarse; // ld, sd (Zilsd)
{1'b1, 1'bz, 1'b0, 1'bz, 16'bzz11zzzzz0z0zzzz}: predecode_rs2_coarse = next_instr[24:20]; // ld, sd (no Zilsd)
{1'b1, 1'bz, 1'bz, 1'bz, 16'bzz0zzzzzzzzzzzzz}: predecode_rs2_coarse = next_instr[24:20]; // (cover remaining 32-bit
{1'b1, 1'bz, 1'bz, 1'bz, 16'bzz10zzzzzzzzzzzz}: predecode_rs2_coarse = next_instr[24:20]; // patterns, without overlap)
{1'b1, 1'bz, 1'bz, 1'bz, 16'bzz11zzzzz1zzzzzz}: predecode_rs2_coarse = next_instr[24:20];
{1'b1, 1'bz, 1'bz, 1'bz, 16'bzz11zzzzz0z1zzzz}: predecode_rs2_coarse = next_instr[24:20];
{1'b0, 1'bz, 1'b1, 1'b1, 16'bzz1zzzzzzzzzzz00}: predecode_rs2_coarse = zclsd_sd_rs2_coarse;
{1'b0, 1'bz, 1'bz, 1'bz, 16'bzz0zzzzzzzzzzz10}: predecode_rs2_coarse = next_instr[6:2]; // c.add, c.swsp
{1'b0, 1'b1, 1'bz, 1'bz, 16'bz01zzzzzzzzzzz10}: predecode_rs2_coarse = zcmp_pushpop_rs2; // cm.push
{1'b0, 1'bz, 1'b1, 1'b1, 16'bz11zzzzzzzzzzz10}: predecode_rs2_coarse = zclsd_sdsp_rs2_coarse;
default: predecode_rs2_coarse = {2'b01, next_instr[4:2]};
endcase
// The "fine" predecode targets those instructions which either:
// - Have an implicit zero-register operand in their expanded form (e.g. c.beqz)
// - Do not have a register operand on that port, but rely on the port being 0
// We don't care about instructions which ignore the reg ports, e.g. ebreak
casez ({|EXTENSION_C, next_instr})
// -> addi rd, x0, imm:
{1'b1, 16'hzzzz, `RVOPC_C_LI}: predecode_rs1_fine = 5'd0;
{1'b1, 16'hzzzz, `RVOPC_C_MV}: begin
if (next_instr[6:2] == 5'd0) begin
// c.jr has rs1 as normal
predecode_rs1_fine = predecode_rs1_coarse;
end else begin
// -> add rd, x0, rs2:
predecode_rs1_fine = 5'd0;
end
end
default: predecode_rs1_fine = predecode_rs1_coarse;
endcase
casez ({|EXTENSION_C, |EXTENSION_ZILSD, |EXTENSION_ZCLSD, next_instr})
{1'b1, 1'bz, 1'bz, 16'hzzzz, `RVOPC_C_BEQZ}: predecode_rs2_fine = 5'd0; // -> beq rs1, x0, label
{1'b1, 1'bz, 1'bz, 16'hzzzz, `RVOPC_C_BNEZ}: predecode_rs2_fine = 5'd0; // -> bne rs1, x0, label
{1'b1, 1'b1, 1'bz, `RVOPC_SD }: predecode_rs2_fine = predecode_rs2_coarse & {5{|next_instr[24:21]}};
{1'b1, 1'b1, 1'b1, 16'hzzzz, `RVOPC_C_SDSP}: predecode_rs2_fine = predecode_rs2_coarse & {5{|next_instr[ 6: 3]}};
default: predecode_rs2_fine = predecode_rs2_coarse;
endcase
if (|EXTENSION_E) begin
predecode_rs1_coarse[4] = 1'b0;
predecode_rs2_coarse[4] = 1'b0;
predecode_rs1_fine[4] = 1'b0;
predecode_rs2_fine[4] = 1'b0;
end
end
// ----------------------------------------------------------------------------
// Instruction decompression
// Instructions are decompressed at the end of stage 1 (fetch data phase). On
// ASIC, where the register file is synthesised with muxes, this puts
// decompression somewhat in parallel with register file read, which uses
// approximately decoded regnums.
generate
if (~|EXTENSION_C) begin: no_decompress
// No decompression; instructions decoded directly from prefetch buffer
always @ (*) begin
cir = {
buf_contents[1 * W_SLOT +: W_BUNDLE],
buf_contents[0 * W_SLOT +: W_BUNDLE]
} | 32'd3;
cir_is_32bit = 1'b1;
cir_invalid_16bit = ~&buf_contents[1:0];
cir_is_uop = 1'b0;
cir_uop_nonfinal = 1'b0;
cir_uop_no_pc_update = 1'b0;
cir_uop_atomic = 1'b0;
end
assign decomp_uop_step = 4'h0;
end else begin: have_decompress
wire decomp_instr_is_32bit;
wire [31:0] decomp_instr_out;
wire decomp_is_uop;
wire decomp_is_final_uop;
wire decomp_uop_no_pc_update;
wire decomp_uop_atomic;
wire decomp_invalid;
wire first_uop = ~|decomp_uop_step;
// Ensure the first uop goes straight through, as it is registered into CIR:
wire uop_stall_non_first = first_uop ? ~|buf_level_next : uop_stall;
// Ensure the uop counter stops at 0 after rolling over once:
wire uop_stall_on_repeat = cir_is_uop && !cir_uop_nonfinal && ~|cir_use;
hazard3_instr_decompress #(
`include "hazard3_config_inst.vh"
) decomp (
.clk (clk),
.rst_n (rst_n),
.instr_in (next_instr),
.instr_is_32bit (decomp_instr_is_32bit),
.instr_out (decomp_instr_out),
.instr_out_is_uop (decomp_is_uop),
.instr_out_is_final_uop (decomp_is_final_uop),
.instr_out_uop_no_pc_update (decomp_uop_no_pc_update),
.instr_out_uop_atomic (decomp_uop_atomic),
.instr_out_uop_stall (uop_stall_non_first || uop_stall_on_repeat),
.instr_out_uop_clear (uop_clear),
.df_uop_step (decomp_uop_step),
.invalid (decomp_invalid)
);
wire cir_clken =
~|cir_vld || (!cir_vld[1] && &buf_contents[1:0]) ||
|cir_use || (|EXTENSION_ZCMP && cir_is_uop && !uop_stall) ||
(|EXTENSION_ZCMP && uop_clear);
wire cir_is_uop_next = decomp_is_uop && |buf_level_next;
wire cir_uop_nonfinal_next = cir_is_uop_next && !decomp_is_final_uop;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
cir <= 32'd3;
cir_is_32bit <= 1'b0;
cir_invalid_16bit <= 1'b0;
cir_is_uop <= 1'b0;
cir_uop_nonfinal <= 1'b0;
cir_uop_no_pc_update <= 1'b0;
cir_uop_atomic <= 1'b0;
end else if (cir_clken) begin
cir <= decomp_instr_out | 32'd3;
cir_is_32bit <= decomp_instr_is_32bit;
cir_invalid_16bit <= decomp_invalid;
cir_is_uop <= |EXTENSION_ZCMP && !uop_clear && cir_is_uop_next;
cir_uop_nonfinal <= |EXTENSION_ZCMP && !uop_clear && cir_uop_nonfinal_next;
cir_uop_no_pc_update <= |EXTENSION_ZCMP && !uop_clear && decomp_uop_no_pc_update;
cir_uop_atomic <= |EXTENSION_ZCMP && !uop_clear && decomp_uop_atomic;
end
end
end
endgenerate
assign cir_raw = {
buf_contents[1 * W_SLOT +: W_BUNDLE],
buf_contents[0 * W_SLOT +: W_BUNDLE]
};
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+478
View File
@@ -0,0 +1,478 @@
/*****************************************************************************\
| Copyright (C) 2021-2023 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Little instructions go in, big instructions come out
`default_nettype none
module hazard3_instr_decompress #(
`include "hazard3_config.vh"
) (
input wire clk,
input wire rst_n,
input wire [31:0] instr_in,
output reg instr_is_32bit,
output reg [31:0] instr_out,
// If instruction is a non-final uop, need to suppress PC update, and null
// the PC offset in the mepc address in stage 3.
output wire instr_out_is_uop,
output wire instr_out_is_final_uop,
output wire instr_out_uop_no_pc_update,
// Indicate instr_out is a uop from the noninterruptible part of a uop
// sequence. If one uop is noninterruptible, all following uops until the
// end of the sequence are also noninterruptible.
output wire instr_out_uop_atomic,
// Current ucode sequence is stalled on downstream execution
input wire instr_out_uop_stall,
input wire instr_out_uop_clear,
// To regnum decoder in frontend
output wire [3:0] df_uop_step,
output reg invalid
);
`include "rv_opcodes.vh"
localparam W_REGADDR = 5;
localparam PASSTHROUGH = ~|EXTENSION_C;
// Long-register formats: cr, ci, css
// Short-register formats: ciw, cl, cs, cb, cj
wire [W_REGADDR-1:0] rd_l = instr_in[11:7];
wire [W_REGADDR-1:0] rs1_l = instr_in[11:7];
wire [W_REGADDR-1:0] rs2_l = instr_in[6:2];
wire [W_REGADDR-1:0] rd_s = {2'b01, instr_in[4:2]};
wire [W_REGADDR-1:0] rs1_s = {2'b01, instr_in[9:7]};
wire [W_REGADDR-1:0] rs2_s = {2'b01, instr_in[4:2]};
// Mapping of cx -> x immediate formats (we are *expanding* instructions, not
// decoding them):
wire [31:0] imm_ci = {
{7{instr_in[12]}},
instr_in[6:2],
20'h00000
};
wire [31:0] imm_cj = {
instr_in[12],
instr_in[8],
instr_in[10:9],
instr_in[6],
instr_in[7],
instr_in[2],
instr_in[11],
instr_in[5:3],
{9{instr_in[12]}},
12'h000
};
wire [31:0] imm_cb ={
{4{instr_in[12]}},
instr_in[6:5],
instr_in[2],
13'h0000,
instr_in[11:10],
instr_in[4:3],
instr_in[12],
7'h00
};
wire [31:0] imm_c_lb = {
10'h0,
instr_in[5],
instr_in[6],
20'h00000
};
wire [31:0] imm_c_lh = {
10'h000,
instr_in[5],
1'b0,
20'h00000
};
function [31:0] rfmt_rd; input [4:0] rd; begin rfmt_rd = {20'h00000, rd, 7'h00}; end endfunction
function [31:0] rfmt_rs1; input [4:0] rs1; begin rfmt_rs1 = {12'h000, rs1, 15'h0000}; end endfunction
function [31:0] rfmt_rs2; input [4:0] rs2; begin rfmt_rs2 = {7'h00, rs2, 20'h00000}; end endfunction
// ----------------------------------------------------------------------------
// Push/pop and friends
// The longest uop sequence is a maximal cm.popretz:
//
// - 13x lw (counter = 0..12)
// - 1x addi to set a0 to zero (counter = 13 ) < atomic section
// - 1x jalr to jump through ra (counter = 14 ) < atomic section
// - 1x addi to adjust sp (counter = 15 ) < atomic section
wire [3:0] uop_ctr;
reg [3:0] uop_ctr_nxt_in_seq;
reg in_uop_seq;
reg uop_no_pc_update;
wire zcmp_is_pushpop = instr_in[12];
wire uop_seq_end = |EXTENSION_ZCMP && (zcmp_is_pushpop ? uop_ctr == 4'hf : uop_ctr[0]);
wire uop_atomic = |EXTENSION_ZCMP && (zcmp_is_pushpop ? uop_ctr >= 4'he : uop_ctr[0]);
wire [3:0] uop_ctr_nxt =
instr_out_uop_clear ? 4'h0 :
instr_out_uop_stall ? uop_ctr : uop_ctr_nxt_in_seq;
assign instr_out_is_uop = in_uop_seq;
assign instr_out_is_final_uop = uop_seq_end;
assign instr_out_uop_atomic = uop_atomic;
assign instr_out_uop_no_pc_update = uop_no_pc_update;
assign df_uop_step = uop_ctr;
// The offset from current sp value to the lowest-addressed saved register, +64.
wire [3:0] zcmp_rlist = instr_in[7:4];
wire [3:0] zcmp_n_regs = zcmp_rlist == 4'hf ? 4'hd : zcmp_rlist - 4'h3;
wire zcmp_rlist_invalid = zcmp_rlist < 4'h4 || (|EXTENSION_E && zcmp_rlist > 4'h6);
wire [11:0] zcmp_stack_adj_base =
zcmp_rlist == 4'hf ? 12'h040 :
zcmp_rlist >= 4'hc ? 12'h030 :
zcmp_rlist >= 4'h8 ? 12'h020 : 12'h010;
wire [11:0] zcmp_stack_adj = zcmp_stack_adj_base + {6'h00, instr_in[3:2], 4'h0};
// Note we perform all load/stores before moving the stack pointer.
wire [11:0] zcmp_stack_lw_offset = -{6'h00, {zcmp_n_regs - uop_ctr}, 2'h0} + zcmp_stack_adj;
wire [11:0] zcmp_stack_sw_offset = -{6'h00, {zcmp_n_regs - uop_ctr}, 2'h0};
wire [4:0] zcmp_ls_reg =
uop_ctr == 4'h0 ? 5'd01 : // ra
uop_ctr == 4'h1 ? 5'd08 : // s0
uop_ctr == 4'h2 ? 5'd09 : // s1
5'd15 + {1'b0, uop_ctr}; // s2-s11 (s2 == x18)
wire [31:0] zcmp_push_sw_instr = `RVOPC_NOZ_SW | rfmt_rs1(5'd2) | rfmt_rs2(zcmp_ls_reg) | {
zcmp_stack_sw_offset[11:5], 13'h0000, zcmp_stack_sw_offset[4:0], 7'h00
};
wire [31:0] zcmp_pop_lw_instr = `RVOPC_NOZ_LW | rfmt_rd(zcmp_ls_reg) | rfmt_rs1(5'd2)| {
zcmp_stack_lw_offset[11:0], 20'h00000
};
wire [31:0] zcmp_push_stack_adj_instr = `RVOPC_NOZ_ADDI | rfmt_rd(5'd2) | rfmt_rs1(5'd2) | {
-zcmp_stack_adj,
20'h00000
};
wire [31:0] zcmp_pop_stack_adj_instr = `RVOPC_NOZ_ADDI | rfmt_rd(5'd2) | rfmt_rs1(5'd2) | {
zcmp_stack_adj,
20'h00000
};
wire [4:0] zcmp_sa01_r1s = {|instr_in[9:8], ~|instr_in[9:8], instr_in[9:7]};
wire [4:0] zcmp_sa01_r2s = {|instr_in[4:3], ~|instr_in[4:3], instr_in[4:2]};
wire zcmp_sa01_invalid = |EXTENSION_E && |{instr_in[9:8], instr_in[4:3]};
// ----------------------------------------------------------------------------
generate
if (PASSTHROUGH) begin: instr_passthrough
always @ (*) begin
instr_is_32bit = 1'b1;
instr_out = instr_in;
invalid = 1'b0;
end
end else begin: instr_decompress
always @ (*) begin
if (instr_in[1:0] == 2'b11) begin
instr_is_32bit = 1'b1;
instr_out = instr_in;
invalid = 1'b0;
in_uop_seq = 1'b0;
uop_no_pc_update = 1'b0;
uop_ctr_nxt_in_seq = uop_ctr;
end else begin
instr_is_32bit = 1'b0;
instr_out = 32'd0;
invalid = 1'b0;
in_uop_seq = 1'b0;
uop_no_pc_update = 1'b0;
uop_ctr_nxt_in_seq = uop_ctr;
casez (instr_in[15:0])
`RVOPC_C_ADDI4SPN: begin
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(rd_s) | rfmt_rs1(5'd2)
| {2'h0, instr_in[10:7], instr_in[12:11], instr_in[5], instr_in[6], 2'b00, 20'h00000};
invalid = ~|instr_in[12:2]; // Always-invalid all-zeroes instruction
end
`RVOPC_C_LW: instr_out = `RVOPC_NOZ_LW | rfmt_rd(rd_s) | rfmt_rs1(rs1_s)
| {5'h00, instr_in[5], instr_in[12:10], instr_in[6], 2'b00, 20'h00000};
`RVOPC_C_SW: instr_out = `RVOPC_NOZ_SW | rfmt_rs2(rs2_s) | rfmt_rs1(rs1_s)
| {5'h00, instr_in[5], instr_in[12], 13'h0000, instr_in[11:10], instr_in[6], 2'b00, 7'h00};
`RVOPC_C_ADDI: instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(rd_l) | rfmt_rs1(rs1_l) | imm_ci;
`RVOPC_C_JAL: instr_out = `RVOPC_NOZ_JAL | rfmt_rd(5'd1) | imm_cj;
`RVOPC_C_J: instr_out = `RVOPC_NOZ_JAL | rfmt_rd(5'd0) | imm_cj;
`RVOPC_C_LI: instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(rd_l) | imm_ci;
`RVOPC_C_LUI: begin
if (rd_l == 5'd2) begin
// addi16sp
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(5'd2) | rfmt_rs1(5'd2) |
{{3{instr_in[12]}}, instr_in[4:3], instr_in[5], instr_in[2], instr_in[6], 24'h000000};
end else begin
instr_out = `RVOPC_NOZ_LUI | rfmt_rd(rd_l) | {{15{instr_in[12]}}, instr_in[6:2], 12'h000};
end
invalid = ~|{instr_in[12], instr_in[6:2]}; // RESERVED if imm == 0
end
`RVOPC_C_SLLI: instr_out = `RVOPC_NOZ_SLLI | rfmt_rd(rs1_l) | rfmt_rs1(rs1_l) | imm_ci;
`RVOPC_C_SRAI: instr_out = `RVOPC_NOZ_SRAI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | imm_ci;
`RVOPC_C_SRLI: instr_out = `RVOPC_NOZ_SRLI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | imm_ci;
`RVOPC_C_ANDI: instr_out = `RVOPC_NOZ_ANDI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | imm_ci;
`RVOPC_C_AND: instr_out = `RVOPC_NOZ_AND | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
`RVOPC_C_OR: instr_out = `RVOPC_NOZ_OR | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
`RVOPC_C_XOR: instr_out = `RVOPC_NOZ_XOR | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
`RVOPC_C_SUB: instr_out = `RVOPC_NOZ_SUB | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
`RVOPC_C_ADD: begin
if (|rs2_l) begin
instr_out = `RVOPC_NOZ_ADD | rfmt_rd(rd_l) | rfmt_rs1(rs1_l) | rfmt_rs2(rs2_l);
end else if (|rs1_l) begin // jalr
instr_out = `RVOPC_NOZ_JALR | rfmt_rd(5'd1) | rfmt_rs1(rs1_l);
end else begin // ebreak
instr_out = `RVOPC_NOZ_EBREAK;
end
end
`RVOPC_C_MV: begin
if (|rs2_l) begin // mv
instr_out = `RVOPC_NOZ_ADD | rfmt_rd(rd_l) | rfmt_rs2(rs2_l);
end else begin // jr
instr_out = `RVOPC_NOZ_JALR | rfmt_rs1(rs1_l);
invalid = ~|rs1_l; // RESERVED
end
end
`RVOPC_C_LWSP: begin
instr_out = `RVOPC_NOZ_LW | rfmt_rd(rd_l) | rfmt_rs1(5'd2) |
{4'h0, instr_in[3:2], instr_in[12], instr_in[6:4], 2'b00, 20'h00000};
invalid = ~|rd_l; // RESERVED
end
`RVOPC_C_SWSP: instr_out = `RVOPC_NOZ_SW | rfmt_rs2(rs2_l) | rfmt_rs1(5'd2)
| {4'h0, instr_in[8:7], instr_in[12], 13'h0000, instr_in[11:9], 2'b00, 7'h00};
`RVOPC_C_BEQZ: instr_out = `RVOPC_NOZ_BEQ | rfmt_rs1(rs1_s) | imm_cb;
`RVOPC_C_BNEZ: instr_out = `RVOPC_NOZ_BNE | rfmt_rs1(rs1_s) | imm_cb;
// Optional Zcb instructions:
`RVOPC_C_LBU: begin
instr_out = `RVOPC_NOZ_LBU | rfmt_rd(rd_s) | rfmt_rs1(rs1_s) | imm_c_lb;
invalid = ~|EXTENSION_ZCB;
end
`RVOPC_C_LHU: begin
instr_out = `RVOPC_NOZ_LHU | rfmt_rd(rd_s) | rfmt_rs1(rs1_s) | imm_c_lh;
invalid = ~|EXTENSION_ZCB;
end
`RVOPC_C_LH: begin
instr_out = `RVOPC_NOZ_LH | rfmt_rd(rd_s) | rfmt_rs1(rs1_s) | imm_c_lh;
invalid = ~|EXTENSION_ZCB;
end
`RVOPC_C_SB: begin
instr_out = `RVOPC_NOZ_SB | rfmt_rs2(rd_s) | rfmt_rs1(rs1_s) | imm_c_lb >> 13;
invalid = ~|EXTENSION_ZCB;
end
`RVOPC_C_SH: begin
instr_out = `RVOPC_NOZ_SH | rfmt_rs2(rd_s) | rfmt_rs1(rs1_s) | imm_c_lh >> 13;
invalid = ~|EXTENSION_ZCB;
end
`RVOPC_C_ZEXT_B: begin
instr_out = `RVOPC_NOZ_ANDI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | 32'h0ff00000;
invalid = ~|EXTENSION_ZCB;
end
`RVOPC_C_SEXT_B: begin
instr_out = `RVOPC_NOZ_SEXT_B | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s);
invalid = ~|EXTENSION_ZCB || ~|EXTENSION_ZBB;
end
`RVOPC_C_ZEXT_H: begin
instr_out = `RVOPC_NOZ_ZEXT_H | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s);
invalid = ~|EXTENSION_ZCB || ~|EXTENSION_ZBB;
end
`RVOPC_C_SEXT_H: begin
instr_out = `RVOPC_NOZ_SEXT_H | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s);
invalid = ~|EXTENSION_ZCB || ~|EXTENSION_ZBB;
end
`RVOPC_C_NOT: begin
instr_out = `RVOPC_NOZ_XORI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | 32'hfff00000;
invalid = ~|EXTENSION_ZCB;
end
`RVOPC_C_MUL: begin
instr_out = `RVOPC_NOZ_MUL | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
invalid = ~|EXTENSION_ZCB || ~|EXTENSION_M;
end
// Optional Zclsd instructions:
`RVOPC_C_LD: begin
instr_out = `RVOPC_NOZ_LD | rfmt_rd(rd_s) | rfmt_rs1(rs1_s)
| {4'h0, instr_in[6:5], instr_in[12:10], 3'b000, 20'h00000};
invalid = ~|EXTENSION_ZILSD || ~|EXTENSION_ZCLSD;
end
`RVOPC_C_SD: begin
instr_out = `RVOPC_NOZ_SD | rfmt_rs2(rs2_s) | rfmt_rs1(rs1_s)
| {4'h0, instr_in[6:5], instr_in[12], 13'h0000, instr_in[11:10], 3'b000, 7'h00};
invalid = ~|EXTENSION_ZILSD || ~|EXTENSION_ZCLSD;
end
`RVOPC_C_LDSP: begin
instr_out = `RVOPC_NOZ_LD | rfmt_rd(rd_l) | rfmt_rs1(5'd2) |
{3'h0, instr_in[4:2], instr_in[12], instr_in[6:5], 3'b000, 20'h00000};
invalid = ~|EXTENSION_ZILSD || ~|EXTENSION_ZCLSD || ~|rd_l; // RESERVED
end
`RVOPC_C_SDSP: begin
instr_out = `RVOPC_NOZ_SD | rfmt_rs2(rs2_l) | rfmt_rs1(5'd2)
| {3'h0, instr_in[9:7], instr_in[12], 13'h0000, instr_in[11:10], 3'b000, 7'h00};
invalid = ~|EXTENSION_ZILSD || ~|EXTENSION_ZCLSD;
end
// Optional Zcmp instructions:
`RVOPC_CM_PUSH: if (~|EXTENSION_ZCMP || zcmp_rlist_invalid) begin
invalid = 1'b1;
end else if (uop_ctr == 4'hf) begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = 4'h0;
instr_out = zcmp_push_stack_adj_instr;
end else begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
instr_out = zcmp_push_sw_instr;
uop_no_pc_update = 1'b1;
if (uop_ctr_nxt_in_seq == zcmp_n_regs) begin
uop_ctr_nxt_in_seq = 4'hf;
end
end
`RVOPC_CM_POP: if (~|EXTENSION_ZCMP || zcmp_rlist_invalid) begin
invalid = 1'b1;
end else if (uop_ctr == 4'hf) begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = 4'h0;
instr_out = zcmp_pop_stack_adj_instr;
end else begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
uop_no_pc_update = 1'b1;
instr_out = zcmp_pop_lw_instr;
if (uop_ctr_nxt_in_seq == zcmp_n_regs) begin
uop_ctr_nxt_in_seq = 4'hf;
end
end
`RVOPC_CM_POPRET: if (~|EXTENSION_ZCMP || zcmp_rlist_invalid) begin
invalid = 1'b1;
end else if (uop_ctr == 4'he) begin
// Note although this is only the first instruction in the uninterruptible sequence,
// we mark this instruction as uninterruptible: there is some special case logic to
// allow this jump to execute without flushing the final stack adjust uop, which can
// cause the wrong exception PC to be sampled if this uop is interrupted.
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
instr_out = `RVOPC_NOZ_JALR | rfmt_rs1(5'd1);
end else if (uop_ctr == 4'hf) begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = 4'h0;
uop_no_pc_update = 1'b1;
instr_out = zcmp_pop_stack_adj_instr;
end else begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
instr_out = zcmp_pop_lw_instr;
uop_no_pc_update = 1'b1;
if (uop_ctr_nxt_in_seq == zcmp_n_regs) begin
uop_ctr_nxt_in_seq = 4'he;
end
end
`RVOPC_CM_POPRETZ: if (~|EXTENSION_ZCMP || zcmp_rlist_invalid) begin
invalid = 1'b1;
end else if (uop_ctr == 4'hd) begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
uop_no_pc_update = 1'b1;
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(5'd10); // li a0, 0
end else if (uop_ctr == 4'he) begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
instr_out = `RVOPC_NOZ_JALR | rfmt_rs1(5'd1);
end else if (uop_ctr == 4'hf) begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = 4'h0;
uop_no_pc_update = 1'b1;
instr_out = zcmp_pop_stack_adj_instr;
end else begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
uop_no_pc_update = 1'b1;
instr_out = zcmp_pop_lw_instr;
if (uop_ctr_nxt_in_seq == zcmp_n_regs) begin
uop_ctr_nxt_in_seq = 4'hd;
end
end
`RVOPC_CM_MVSA01: if (~|EXTENSION_ZCMP || zcmp_sa01_invalid) begin
invalid = 1'b1;
end else if (uop_ctr == 4'h0) begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
uop_no_pc_update = 1'b1;
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(zcmp_sa01_r1s) | rfmt_rs1(5'd10);
end else begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = 4'h0;
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(zcmp_sa01_r2s) | rfmt_rs1(5'd11);
end
`RVOPC_CM_MVA01S: if (~|EXTENSION_ZCMP || zcmp_sa01_invalid) begin
invalid = 1'b1;
end else if (uop_ctr == 4'h0) begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
uop_no_pc_update = 1'b1;
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(5'd10) | rfmt_rs1(zcmp_sa01_r1s);
end else begin
in_uop_seq = 1'b1;
uop_ctr_nxt_in_seq = 4'h0;
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(5'd11) | rfmt_rs1(zcmp_sa01_r2s);
end
default: invalid = 1'b1;
endcase
end
end
end
endgenerate
generate
if (EXTENSION_ZCMP) begin: have_uop_ctr
reg [3:0] uop_ctr_r;
assign uop_ctr = uop_ctr_r;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
uop_ctr_r <= 4'h0;
end else begin
uop_ctr_r <= uop_ctr_nxt;
`ifdef HAZARD3_ASSERTIONS
assert(in_uop_seq || uop_ctr_r == 4'h0);
assert(in_uop_seq || zcmp_ls_reg == 5'h01);
assert(in_uop_seq || !uop_atomic);
assert(in_uop_seq || !uop_no_pc_update);
if (uop_seq_end) begin
assert(in_uop_seq);
assert(instr_out_uop_stall || uop_ctr_nxt == 4'h0);
end
`endif
end
end
end else begin: no_uop_ctr
assign uop_ctr = 4'h0;
end
endgenerate
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+327
View File
@@ -0,0 +1,327 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
`default_nettype none
// Hazard3 interrupt controller. Support for up to 512 external interrupt
// lines, with up to 16 levels of preemption.
module hazard3_irq_ctrl #(
`include "hazard3_config.vh"
) (
input wire clk,
input wire clk_always_on,
input wire rst_n,
// CSR interface
input wire [11:0] addr,
input wire [1:0] wtype,
input wire wen_m_mode,
input wire ren_m_mode,
input wire [W_DATA-1:0] wdata_raw,
input wire [W_DATA-1:0] wdata,
output reg [W_DATA-1:0] rdata,
// Trap entry/exit signals for context update
input wire trapreg_update_enter,
input wire trapreg_update_exit,
input wire trap_entry_is_eirq,
// Interface for clearing and saving mie.mtie/msie via meicontext
output wire meicontext_clearts,
input wire mie_mtie,
input wire mie_msie,
// External IRQ inputs:
input wire [NUM_IRQS-1:0] irq,
// mip.meip:
output wire external_irq_pending
);
`include "hazard3_ops.vh"
`include "hazard3_csr_addr.vh"
localparam MAX_IRQS = 512;
localparam [3:0] IRQ_PRIORITY_MASK = ~(4'hf >> IRQ_PRIORITY_BITS);
localparam W_IRQ_INDEX = $clog2(MAX_IRQS);
// ----------------------------------------------------------------------------
// IRQ input flops
// Register external IRQ signals (mainly to avoid a through-path from IRQs to
// bus request signals). Always clocked, as it's used to generate a wakeup.
// Input registers can be removed on a per-IRQ basis, but this should be done
// with care as it does create a through-path from the IRQ to the bus.
wire [NUM_IRQS-1:0] irq_r;
genvar g;
generate
for (g = 0; g < NUM_IRQS; g = g + 1) begin: irq_reg_loop
if (IRQ_INPUT_BYPASS[g]) begin: no_reg
assign irq_r[g] = irq[g];
end else begin: have_reg
reg q;
always @ (posedge clk_always_on or negedge rst_n) begin
if (!rst_n) begin
q <= 1'b0;
end else begin
q <= irq[g];
end
end
assign irq_r[g] = q;
end
end
endgenerate
// ----------------------------------------------------------------------------
// CSR write
// Assigned later:
wire [W_IRQ_INDEX-1:0] meinext_irq;
wire meinext_noirq;
reg [3:0] eirq_highest_priority;
// Interrupt array registers:
reg [NUM_IRQS-1:0] meiea;
reg [NUM_IRQS-1:0] meifa;
reg [4*NUM_IRQS-1:0] meipra;
// Padded vectors for CSR readout
wire [MAX_IRQS-1:0] meiea_rdata = {{MAX_IRQS-NUM_IRQS{1'b0}}, meiea};
wire [MAX_IRQS-1:0] meifa_rdata = {{MAX_IRQS-NUM_IRQS{1'b0}}, meifa};
wire [4*MAX_IRQS-1:0] meipra_rdata = {{4*(MAX_IRQS-NUM_IRQS){1'b0}}, meipra};
always @ (posedge clk or negedge rst_n) begin: update_irq_reg_arrays
reg signed [31:0] i;
if (!rst_n) begin
meiea <= {NUM_IRQS{1'b0}};
meifa <= {NUM_IRQS{1'b0}};
meipra <= {4*NUM_IRQS{1'b0}};
end else begin
for (i = 0; i < NUM_IRQS; i = i + 1) begin
// CSR write update. Note raw wdata is used for array indexing --
// necessary for correctness, and also avoid a loop with rdata.
if (wen_m_mode && addr == MEIEA && $signed(wdata_raw[4:0]) == i[W_IRQ_INDEX-1:4]) begin
meiea[i] <= wdata[16 + (i % 16)];
end
if (wen_m_mode && addr == MEIFA && $signed(wdata_raw[4:0]) == i[W_IRQ_INDEX-1:4]) begin
meifa[i] <= wdata[16 + (i % 16)];
end
if (wen_m_mode && addr == MEIPRA && $signed(wdata_raw[6:0]) == i[W_IRQ_INDEX-1:2]) begin
meipra[4 * i +: 4] <= wdata[16 + 4 * (i % 4) +: 4] & IRQ_PRIORITY_MASK;
end
// Clear IRQ force when the corresponding IRQ is sampled from meinext
// (so that an IRQ can be posted *once* without modifying the ISR source)
if (meinext_irq == i[W_IRQ_INDEX-1:0] && ren_m_mode && addr == MEINEXT && !meinext_noirq) begin
meifa[i[$clog2(NUM_IRQS)-1:0]] <= 1'b0;
end
end
end
end
reg [3:0] meicontext_pppreempt;
reg [3:0] meicontext_ppreempt;
reg [4:0] meicontext_preempt;
reg meicontext_noirq;
reg [W_IRQ_INDEX-1:0] meicontext_irq;
reg meicontext_mreteirq;
wire [4:0] preempt_level_next = meinext_noirq ? 5'h10 : (
(5'd1 << (4 - IRQ_PRIORITY_BITS)) + {1'b0, eirq_highest_priority}
);
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
meicontext_pppreempt <= 4'h0;
meicontext_ppreempt <= 4'h0;
meicontext_preempt <= 5'h0;
meicontext_noirq <= 1'b1;
meicontext_irq <= {W_IRQ_INDEX{1'b0}};
meicontext_mreteirq <= 1'b0;
end else if (trapreg_update_enter) begin
if (trap_entry_is_eirq) begin
// Priority save. Note the MSB of preempt needn't be saved since,
// when it is set, preemption is impossible, so we won't be here.
meicontext_pppreempt <= meicontext_ppreempt & IRQ_PRIORITY_MASK;
meicontext_ppreempt <= meicontext_preempt[3:0] & IRQ_PRIORITY_MASK;
// Setting preempt isn't strictly necessary, since an updating read
// of meinext ought to be performed before re-enabling IRQs via
// mstatus.mie, but it seems the least surprising thing to do:
meicontext_preempt <= preempt_level_next & {1'b1, IRQ_PRIORITY_MASK};
meicontext_mreteirq <= 1'b1;
end else begin
meicontext_mreteirq <= 1'b0;
end
end else if (trapreg_update_exit) begin
meicontext_mreteirq <= 1'b0;
if (meicontext_mreteirq) begin
// Priority restore
meicontext_pppreempt <= 4'h0;
meicontext_ppreempt <= meicontext_pppreempt & IRQ_PRIORITY_MASK;
meicontext_preempt <= {1'b0, meicontext_ppreempt & IRQ_PRIORITY_MASK};
end
end else if (wen_m_mode && addr == MEICONTEXT) begin
meicontext_pppreempt <= wdata[31:28] & IRQ_PRIORITY_MASK;
meicontext_ppreempt <= wdata[27:24] & IRQ_PRIORITY_MASK;
meicontext_preempt <= wdata[20:16] & {1'b1, IRQ_PRIORITY_MASK};
meicontext_noirq <= wdata[15];
meicontext_irq <= wdata[12:4];
meicontext_mreteirq <= wdata[0];
end else if (wen_m_mode && addr == MEINEXT && wdata[0]) begin
// Interrupt has been sampled, with the update request set, so update
// the context (including preemption level) appropriately.
meicontext_preempt <= preempt_level_next & {1'b1, IRQ_PRIORITY_MASK};
meicontext_noirq <= meinext_noirq;
meicontext_irq <= meinext_irq;
end
end
assign meicontext_clearts = wen_m_mode && wtype != CSR_WTYPE_C && addr == MEICONTEXT && wdata_raw[1];
// ----------------------------------------------------------------------------
// External interrupt logic
// Trap request is asserted when there is an interrupt at or above our current
// preemption level. meinext displays interrupts at or above our *previous*
// preemption level: this masking helps avoid re-taking IRQs in frames that you
// have preempted.
wire [NUM_IRQS-1:0] meipa = irq_r | meifa;
wire [MAX_IRQS-1:0] meipa_rdata = {{MAX_IRQS-NUM_IRQS{1'b0}}, meipa};
reg [NUM_IRQS-1:0] eirq_active_above_preempt;
reg [NUM_IRQS-1:0] eirq_active_above_ppreempt;
always @ (*) begin: eirq_compare
integer i;
for (i = 0; i < NUM_IRQS; i = i + 1) begin
eirq_active_above_preempt[i] = meipa[i] && meiea[i] && {1'b0, meipra[i * 4 +: 4]} >= meicontext_preempt;
eirq_active_above_ppreempt[i] = meipa[i] && meiea[i] && meipra[i * 4 +: 4] >= meicontext_ppreempt;
end
end
assign external_irq_pending = |eirq_active_above_preempt;
assign meinext_noirq = ~|eirq_active_above_ppreempt;
// Two things remaining to calculate:
//
// - What is the IRQ number of the highest-priority pending IRQ that is above
// meicontext.ppreempt
// - What is the priority of that IRQ
//
// In the second case we can relax the calculation to ignore ppreempt, since it
// only needs to be valid if such an IRQ exists. Currently we choose to reuse
// the same priority selector (possibly longer critpath while saving area), but
// we could use a second priority selector that ignores ppreempt masking.
wire [NUM_IRQS-1:0] highest_eirq_onehot;
wire [W_IRQ_INDEX-1:0] meinext_irq_unmasked;
hazard3_onehot_priority_dynamic #(
.W_REQ (NUM_IRQS),
.N_PRIORITIES (16),
.PRIORITY_HIGHEST_WINS (1),
.TIEBREAK_HIGHEST_WINS (0)
) eirq_priority_u (
.pri (meipra[4*NUM_IRQS-1:0] & {NUM_IRQS{IRQ_PRIORITY_MASK}}),
.req (eirq_active_above_ppreempt),
.gnt (highest_eirq_onehot)
);
always @ (*) begin: get_highest_eirq_priority
integer i;
eirq_highest_priority = 4'h0;
for (i = 0; i < NUM_IRQS; i = i + 1) begin
eirq_highest_priority = eirq_highest_priority | (
meipra[4 * i +: 4] & {4{highest_eirq_onehot[i]}}
);
end
end
wire [$clog2(NUM_IRQS)-1:0] meinext_irq_unmasked_nopad;
hazard3_onehot_encode #(
.W_REQ (NUM_IRQS)
) eirq_encode_u (
.req (highest_eirq_onehot),
.gnt (meinext_irq_unmasked_nopad)
);
generate
if ($clog2(NUM_IRQS) == $clog2(MAX_IRQS)) begin: encode_eirq_no_padding
assign meinext_irq_unmasked = meinext_irq_unmasked_nopad;
end else begin: encode_eirq_padded
assign meinext_irq_unmasked = {
{$clog2(MAX_IRQS) - $clog2(NUM_IRQS){1'b0}},
meinext_irq_unmasked_nopad
};
end
endgenerate
// It is unnecessary to mask meinext_irq based on meinext_noirq because:
// - The value of the CSR field is unimportant when noirq is set
// - There are no IRQ inputs to the priority selector when there
// are no IRQs, so result is already 0.
assign meinext_irq = meinext_irq_unmasked;
// ----------------------------------------------------------------------------
// CSR read
always @ (*) begin
rdata = {W_DATA{1'b0}};
case (addr)
MEIEA: rdata = {
meiea_rdata[wdata_raw[4:0] * 16 +: 16],
16'h0
};
MEIPA: rdata = {
meipa_rdata[wdata_raw[4:0] * 16 +: 16],
16'h0
};
MEIFA: rdata = {
meifa_rdata[wdata_raw[4:0] * 16 +: 16],
16'h0
};
MEIPRA: rdata = {
meipra_rdata[wdata_raw[6:0] * 16 +: 16],
16'h0
};
MEINEXT: rdata = {
meinext_noirq,
20'h0,
meinext_irq,
2'h0
};
MEICONTEXT: rdata = {
meicontext_pppreempt,
meicontext_ppreempt,
3'h0,
meicontext_preempt,
meicontext_noirq,
2'h0,
meicontext_irq,
mie_mtie && meicontext_clearts,
mie_msie && meicontext_clearts,
1'b0,
meicontext_mreteirq
};
default: rdata = {W_DATA{1'b0}};
endcase
end
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+124
View File
@@ -0,0 +1,124 @@
/*****************************************************************************\
| Copyright (C) 2021 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// ALU operation selectors
localparam ALUOP_ADD = 6'h00;
localparam ALUOP_SUB = 6'h01;
localparam ALUOP_LT = 6'h02;
localparam ALUOP_LTU = 6'h04;
localparam ALUOP_AND = 6'h06;
localparam ALUOP_OR = 6'h07;
localparam ALUOP_XOR = 6'h08;
localparam ALUOP_SRL = 6'h09;
localparam ALUOP_SRA = 6'h0a;
localparam ALUOP_SLL = 6'h0b;
localparam ALUOP_MULDIV = 6'h0c;
localparam ALUOP_RS2 = 6'h0d; // differs from AND/OR/XOR in [1:0]
// Bitmanip ALU operations (some also used by AMOs):
localparam ALUOP_SHXADD = 6'h20;
localparam ALUOP_CLZ = 6'h23;
localparam ALUOP_CPOP = 6'h24;
localparam ALUOP_CTZ = 6'h25;
localparam ALUOP_ANDN = 6'h26; // Same LSBs as non-inverted
localparam ALUOP_ORN = 6'h27; // Same LSBs as non-inverted
localparam ALUOP_XNOR = 6'h28; // Same LSBs as non-inverted
localparam ALUOP_MAX = 6'h29;
localparam ALUOP_MAXU = 6'h2a;
localparam ALUOP_MIN = 6'h2b;
localparam ALUOP_MINU = 6'h2c;
localparam ALUOP_ORC_B = 6'h2d;
localparam ALUOP_REV8 = 6'h2e;
localparam ALUOP_ROL = 6'h2f;
localparam ALUOP_ROR = 6'h30;
localparam ALUOP_SEXT_B = 6'h31;
localparam ALUOP_SEXT_H = 6'h32;
localparam ALUOP_ZEXT_H = 6'h33;
localparam ALUOP_CLMUL = 6'h34;
localparam ALUOP_BCLR = 6'h35;
localparam ALUOP_BEXT = 6'h36;
localparam ALUOP_BINV = 6'h37;
localparam ALUOP_BSET = 6'h38;
localparam ALUOP_PACK = 6'h39;
localparam ALUOP_PACKH = 6'h3a;
localparam ALUOP_BREV8 = 6'h3b;
localparam ALUOP_ZIP = 6'h3c;
localparam ALUOP_UNZIP = 6'h3d;
localparam ALUOP_BEXTM = 6'h3e;
localparam ALUOP_XPERM = 6'h3f;
// Parameters to control ALU input muxes. Bypass mux paths are
// controlled by X, so D has no parameters to choose these.
localparam ALUSRCA_RS1 = 1'h0;
localparam ALUSRCA_PC = 1'h1;
localparam ALUSRCB_RS2 = 1'h0;
localparam ALUSRCB_IMM = 1'h1;
localparam MEMOP_LW = 5'h00;
localparam MEMOP_LH = 5'h01;
localparam MEMOP_LB = 5'h02;
localparam MEMOP_LHU = 5'h03;
localparam MEMOP_LBU = 5'h04;
localparam MEMOP_SW = 5'h05;
localparam MEMOP_SH = 5'h06;
localparam MEMOP_SB = 5'h07;
localparam MEMOP_LR_W = 5'h08;
localparam MEMOP_SC_W = 5'h09;
localparam MEMOP_AMO = 5'h0a;
localparam MEMOP_NONE = 5'h10;
localparam BCOND_NEVER = 2'h0;
localparam BCOND_ALWAYS = 2'h1;
localparam BCOND_ZERO = 2'h2;
localparam BCOND_NZERO = 2'h3;
// CSR access types
localparam CSR_WTYPE_W = 2'h0;
localparam CSR_WTYPE_S = 2'h1;
localparam CSR_WTYPE_C = 2'h2;
// Exceptional condition signals which travel alongside (or instead of)
// instructions in the pipeline. These are speculative and can be flushed on
// e.g. branch mispredict. These mostly align with mcause values.
localparam EXCEPT_NONE = 4'hf;
localparam EXCEPT_INSTR_MISALIGN = 4'h0;
localparam EXCEPT_INSTR_FAULT = 4'h1;
localparam EXCEPT_INSTR_ILLEGAL = 4'h2;
localparam EXCEPT_EBREAK = 4'h3;
localparam EXCEPT_LOAD_ALIGN = 4'h4;
localparam EXCEPT_LOAD_FAULT = 4'h5;
localparam EXCEPT_STORE_ALIGN = 4'h6;
localparam EXCEPT_STORE_FAULT = 4'h7;
localparam EXCEPT_ECALL_U = 4'h8;
// MRET, Return from M-mode: not really an exception, but handled like one
localparam EXCEPT_MRET = 4'ha;
localparam EXCEPT_ECALL_M = 4'hb;
// spare: c
// spare: d
// REFETCH: flush and refetch sequentially-following instructions, e.g. on
// executing fence.i. Jumps from stage 3 to get ordering against L/S dphase.
localparam EXCEPT_REFETCH = 4'he;
// Operations for M extension (these are just instr[14:12])
localparam M_OP_MUL = 3'h0;
localparam M_OP_MULH = 3'h1;
localparam M_OP_MULHSU = 3'h2;
localparam M_OP_MULHU = 3'h3;
localparam M_OP_DIV = 3'h4;
localparam M_OP_DIVU = 3'h5;
localparam M_OP_REM = 3'h6;
localparam M_OP_REMU = 3'h7;
+380
View File
@@ -0,0 +1,380 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
`default_nettype none
// Physical memory protection unit
module hazard3_pmp #(
`include "hazard3_config.vh"
) (
input wire clk,
input wire rst_n,
// Config interface passed through CSR block
input wire [11:0] cfg_addr,
input wire cfg_wen,
input wire [W_DATA-1:0] cfg_wdata,
output reg [W_DATA-1:0] cfg_rdata,
// Fetch address query
input wire [W_ADDR-1:0] i_addr,
input wire i_m_mode,
output wire i_kill,
// Load/store address query
input wire [W_ADDR-1:0] d_addr,
// Broken out separately for carry-save:
input wire [W_ADDR-1:0] d_addr_addend_rs1,
input wire [W_ADDR-1:0] d_addr_addend_imm,
input wire [W_ADDR-1:0] d_addr_addend_lspair_offs,
input wire d_m_mode,
input wire d_write,
output wire d_kill
);
localparam PMP_A_NAPOT = 2'b11;
localparam PMP_A_NA4 = 2'b10;
localparam PMP_A_TOR = 2'b01;
localparam PMP_A_OFF = 2'b00;
// Which values are supported in A field (unsupported are mapped to OFF):
localparam [3:0] PMP_A_SUPPORTED = {
|PMP_MATCH_NAPOT,
|PMP_MATCH_NAPOT && PMP_GRAIN == 0,
|PMP_MATCH_TOR,
1'b1
};
`include "hazard3_csr_addr.vh"
generate
if (PMP_REGIONS == 0) begin: no_pmp
// This should already be stubbed out in core.v, but use a generate here too
// so that we don't get a warning for elaborating this module with a region
// count of 0.
always @ (*) cfg_rdata = {W_DATA{1'b0}};
assign i_kill = 1'b0;
assign d_kill = 1'b0;
end else begin: have_pmp
// ----------------------------------------------------------------------------
// Config registers and read/write interface
// Whether a region's configuration is writable; this is non-trivial when TOR
// is supported because locking region i + 1 can also lock region i.
wire [PMP_REGIONS-1:0] region_locked;
reg [PMP_REGIONS-1:0] pmpcfg_l;
reg [1:0] pmpcfg_a [0:PMP_REGIONS-1];
reg [PMP_REGIONS-1:0] pmpcfg_x;
reg [PMP_REGIONS-1:0] pmpcfg_w;
reg [PMP_REGIONS-1:0] pmpcfg_r;
// Address register contains bits 33:2 of the address (to support 16 GiB
// physical address space). We don't implement bits 33 or 32.
reg [W_ADDR-3:0] pmpaddr [0:PMP_REGIONS-1];
// Hazard3 extension for applying PMP regions to M-mode without locking.
// Different from ePMP mseccfg.rlb: low-numbered regions may be locked for
// security reasons, but higher-numbered regions should stll be available for
// other purposes e.g. stack guarding, peripheral emulation
reg [PMP_REGIONS-1:0] pmpcfg_m;
always @ (posedge clk or negedge rst_n) begin: cfg_update
reg signed [31:0] i;
if (!rst_n) begin
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
pmpcfg_l[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 7] : 1'b0;
pmpcfg_a[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 3 +: 2] : 2'h0;
pmpcfg_x[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 2] : 1'b0;
pmpcfg_w[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 1] : 1'b0;
pmpcfg_r[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 0] : 1'b0;
pmpaddr[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_ADDR[32 * i +: 30] :
PMP_GRAIN > 1 ? ~(~30'h0 << (PMP_GRAIN - 1)) : 30'h0;
end
pmpcfg_m <= {PMP_REGIONS{1'b0}};
end else if (cfg_wen) begin
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
if (cfg_addr == PMPCFG0 + i[13:2] && !region_locked[i]) begin
if (PMP_HARDWIRED[i]) begin
// Keep tied to hardwired value (but still make the "register" sensitive to clk)
pmpcfg_l[i] <= PMP_HARDWIRED_CFG[8 * i + 7];
pmpcfg_a[i] <= PMP_HARDWIRED_CFG[8 * i + 3 +: 2];
pmpcfg_x[i] <= PMP_HARDWIRED_CFG[8 * i + 2];
pmpcfg_w[i] <= PMP_HARDWIRED_CFG[8 * i + 1];
pmpcfg_r[i] <= PMP_HARDWIRED_CFG[8 * i + 0];
pmpaddr[i] <= PMP_HARDWIRED_ADDR[32 * i +: 30];
end else begin
pmpcfg_l[i] <= cfg_wdata[i % 4 * 8 + 7];
pmpcfg_x[i] <= cfg_wdata[i % 4 * 8 + 2];
pmpcfg_w[i] <= cfg_wdata[i % 4 * 8 + 1];
pmpcfg_r[i] <= cfg_wdata[i % 4 * 8 + 0];
// Unsupported A values are mapped to OFF (it's a WARL field).
pmpcfg_a[i] <= PMP_A_SUPPORTED[cfg_wdata[i % 4 * 8 + 3 +: 2]] ?
cfg_wdata[i % 4 * 8 + 3 +: 2] : PMP_A_OFF;
end
end
if (cfg_addr == PMPADDR0 + i[11:0] && !region_locked[i]) begin
// This implements one bit too many when G > 0 and only
// PMP_MATCH_TOR is enabled, however that bit is ignored for
// both rdata and address matching, so should be trimmed.
if (PMP_GRAIN > 1) begin
pmpaddr[i] <= cfg_wdata[W_ADDR-3:0] | ~(~30'h0 << (PMP_GRAIN - 1));
end else begin
pmpaddr[i] <= cfg_wdata[W_ADDR-3:0];
end
end
end
if (cfg_addr == PMPCFGM0) begin
pmpcfg_m <= cfg_wdata[PMP_REGIONS-1:0] & ~PMP_HARDWIRED & {PMP_REGIONS{|EXTENSION_XH3PMPM}};
end
end
end
always @ (*) begin: cfg_read
reg signed [31:0] i;
cfg_rdata = {W_DATA{1'b0}};
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
if (cfg_addr == PMPCFG0 + i[13:2]) begin
cfg_rdata[i % 4 * 8 +: 8] = {
pmpcfg_l[i],
2'b00,
pmpcfg_a[i],
pmpcfg_x[i],
pmpcfg_w[i],
pmpcfg_r[i]
};
end else if (cfg_addr == PMPADDR0 + i[11:0]) begin
if (PMP_GRAIN >= 2 && pmpcfg_a[i][1]) begin
// Bits G-2:0 read back as all-ones when A is NA4 or NAPOT.
cfg_rdata[W_ADDR-3:0] = pmpaddr[i] | ~({W_ADDR-2{1'b1}} << (PMP_GRAIN - 1));
end else if (PMP_GRAIN >= 1 && !pmpcfg_a[i][1]) begin
// Bits G-1:0 read back as all-zeroes when A is OFF or TOR.
cfg_rdata[W_ADDR-3:0] = pmpaddr[i] & ({W_ADDR-2{1'b1}} << PMP_GRAIN);
end else begin
cfg_rdata[W_ADDR-3:0] = pmpaddr[i];
end
end
end
if (cfg_addr == PMPCFGM0) begin
cfg_rdata = {{32-PMP_REGIONS{1'b0}}, pmpcfg_m} & {32{|EXTENSION_XH3PMPM}};
end
end
// ----------------------------------------------------------------------------
// Region locking rules
reg [PMP_REGIONS-1:0] pmp_region_is_tor;
always @ (*) begin: check_region_is_tor
integer i;
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
pmp_region_is_tor[i] = PMP_MATCH_TOR && pmpcfg_a[i] == PMP_A_TOR;
end
end
assign region_locked = pmpcfg_l | ((pmpcfg_l & pmp_region_is_tor) >> 1);
// ----------------------------------------------------------------------------
// Match addresses against regions
wire [PMP_REGIONS-1:0] d_match_napot;
wire [PMP_REGIONS-1:0] i_match_napot;
wire [PMP_REGIONS-1:0] d_match_tor;
wire [PMP_REGIONS-1:0] i_match_tor;
if (PMP_MATCH_NAPOT != 0) begin: have_napot
reg [PMP_REGIONS-1:0] d_match_napot_r;
reg [PMP_REGIONS-1:0] i_match_napot_r;
assign d_match_napot = d_match_napot_r;
assign i_match_napot = i_match_napot_r;
// Decode PMPCFGx.A and PMPADDRx into a 32-bit address mask and address
reg [W_ADDR-1:0] match_mask [0:PMP_REGIONS-1];
reg [W_ADDR-1:0] match_addr [0:PMP_REGIONS-1];
// Encoding: (noting ADDR is a 4-byte address, not a word address):
// CFG.A | ADDR | Region size
// ------+----------+------------
// NA4 | y..yyyyy | 4 bytes
// NAPOT | y..yyyy0 | 8 bytes
// NAPOT | y..yyy01 | 16 bytes
// NAPOT | y..yy011 | 32 bytes
// NAPOT | y..y0111 | 64 bytes
// etc.
//
// So, with the exception of NA4, the rule is to check all bits more
// significant than the least-significant 0 bit.
always @ (*) begin: decode_match_mask_addr
integer i, j;
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
if (!pmpcfg_a[i][0]) begin
match_mask[i] = {{W_ADDR-2{1'b1}}, 2'b00};
end else begin
// Bits 1:0 are always 0. Bit 2 is 0 because NAPOT is at least 8 bytes.
match_mask[i] = {W_ADDR{1'b0}};
for (j = 3; j < W_ADDR; j = j + 1) begin
match_mask[i][j] = match_mask[i][j - 1] || !pmpaddr[i][j - 3];
end
end
match_addr[i] = {pmpaddr[i], 2'b00} & match_mask[i];
end
end
// We check only the least-addressed byte of each access. See later
// comments for an argument as to why this is sufficient.
always @ (*) begin: check_d_match
integer i;
for (i = PMP_REGIONS - 1; i >= 0; i = i - 1) begin
d_match_napot_r[i] = pmpcfg_a[i][1] &&
(d_addr & match_mask[i]) == match_addr[i];
i_match_napot_r[i] = pmpcfg_a[i][1] &&
(i_addr & match_mask[i]) == match_addr[i];
end
end
end else begin: no_napot
assign d_match_napot = {PMP_REGIONS{1'b0}};
assign i_match_napot = {PMP_REGIONS{1'b0}};
end
if (PMP_MATCH_TOR != 0) begin: have_tor
reg [PMP_REGIONS-1:0] d_match_tor_r;
reg [PMP_REGIONS-1:0] i_match_tor_r;
reg [W_ADDR-1:0] watermark [0:PMP_REGIONS-1];
reg [PMP_REGIONS-1:0] d_lt;
reg [PMP_REGIONS-1:0] i_lt;
assign d_match_tor = d_match_tor_r;
assign i_match_tor = i_match_tor_r;
always @ (*) begin: compare
integer i;
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
watermark[i] = {
pmpaddr[i][W_ADDR-3:0] & (~30'h0 << PMP_GRAIN),
2'b00
};
// Bring terms in separately to try to encourage adder merging
d_lt[i] = (
d_addr_addend_rs1 + d_addr_addend_imm + d_addr_addend_lspair_offs
) < watermark[i];
i_lt[i] = i_addr < watermark[i];
end
end
wire [PMP_REGIONS-1:0] d_prev_ge = ~(d_lt << 1);
wire [PMP_REGIONS-1:0] i_prev_ge = ~(i_lt << 1);
always @ (*) begin: match
integer i;
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
d_match_tor_r[i] = d_lt[i] && d_prev_ge[i] && pmpcfg_a[i] == PMP_A_TOR;
i_match_tor_r[i] = i_lt[i] && i_prev_ge[i] && pmpcfg_a[i] == PMP_A_TOR;
end
end
end else begin: no_tor
assign d_match_tor = {PMP_REGIONS{1'b0}};
assign i_match_tor = {PMP_REGIONS{1'b0}};
end
// ----------------------------------------------------------------------------
// Decode permissions from matches
// For load/stores we assume any non-naturally-aligned transfers trigger a
// misaligned load/store/AMO exception, so we only need to decode the PMP
// attribute for the first byte of the access. Note the spec gives us freedom
// to report *either* a load/store/AMO access fault (mcause = 5, 7) or a
// load/store/AMO alignment fault (mcause = 4, 6), in the case that both
// happen, and we choose alignment fault in this case.
reg d_m; // Hazard3 extension (M-mode without locking)
reg d_l;
reg d_r;
reg d_w;
always @ (*) begin: check_d_match
integer i;
d_m = 1'b0;
d_l = 1'b0;
d_r = 1'b0;
d_w = 1'b0;
// Lowest-numbered match wins, so work down from the top. This should be
// inferred as a priority mux structure (cascade mux).
for (i = PMP_REGIONS - 1; i >= 0; i = i - 1) begin
if (d_match_napot[i] || d_match_tor[i]) begin
d_m = pmpcfg_m[i];
d_l = pmpcfg_l[i];
d_r = pmpcfg_r[i];
d_w = pmpcfg_w[i];
end
end
end
// Instructions work similarly because we check *fetches*, not instructions.
// Fetch is always word-sized word-aligned. The spec permits this:
//
// "On some implementations, misaligned loads, stores, and instruction fetches
// may also be decomposed into multiple accesses, some of which may succeed
// before an access-fault exception occurs."
//
// Hazard3 separately checks the naturally-aligned fetches that occur in the
// course of fetching a non-naturally-aligned instruction. This means
// instruction fetch spanning two different regions which both grant X
// permission *is* permitted, unlike the RP2350 version of Hazard3.
reg i_m; // Hazard3 extension (M-mode without locking)
reg i_l;
reg i_x;
always @ (*) begin: check_i_match
integer i;
i_m = 1'b0;
i_l = 1'b0;
i_x = 1'b0;
for (i = PMP_REGIONS - 1; i >= 0; i = i - 1) begin
if (i_match_napot[i] || i_match_tor[i]) begin
i_m = pmpcfg_m[i];
i_l = pmpcfg_l[i];
i_x = pmpcfg_x[i];
end
end
end
// ----------------------------------------------------------------------------
// Access rules
// M-mode gets to ignore protections, unless the lock or M-mode bit is set.
assign d_kill = (!d_m_mode || d_l || d_m) && (
(!d_write && !d_r) ||
( d_write && !d_w)
);
assign i_kill = (!i_m_mode || i_l || i_m) && !i_x;
end
endgenerate
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+173
View File
@@ -0,0 +1,173 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
`default_nettype none
// Wake/sleep (power) state machine for Hazard3
module hazard3_power_ctrl #(
`include "hazard3_config.vh"
) (
input wire clk_always_on,
input wire rst_n,
// 4-phase (Gray code) req/ack handshake for requesting and releasing
// power+clock enable on non-processor hardware, e.g. the bus fabric. This
// can also be used for an external controller to gate the processor's clk
// input, rather than the clk_en signal below.
output reg pwrup_req,
input wire pwrup_ack,
// Top-level clock enable for an optional clock gate on the processor's clk
// input (but not clk_always_on, which clocks this module and the IRQ input
// flops). This allows the processor to clock-gate when sleeping. It's
// acceptable for the clock gate cell to have one cycle of delay when
// clk_en changes.
output reg clk_en,
// Power state controls from CSRs
input wire allow_clkgate,
input wire allow_power_down,
input wire allow_sleep_on_block,
// Signal from frontend that it has stalled against the WFI pipeline
// stall, and we are now clear to enter a deep sleep state
input wire frontend_pwrdown_ok,
input wire sleeping_on_wfi,
input wire wfi_wakeup_req,
input wire sleeping_on_block,
input wire block_wakeup_req_pulse,
output reg stall_release
);
// ----------------------------------------------------------------------------
// Wake/sleep state machine
localparam W_STATE = 2;
localparam S_AWAKE = 2'h0;
localparam S_ENTER_ASLEEP = 2'h1;
localparam S_ASLEEP = 2'h2;
localparam S_ENTER_AWAKE = 2'h3;
reg [W_STATE-1:0] state;
reg block_wakeup_req;
wire active_wake_req =
(sleeping_on_block && (block_wakeup_req || wfi_wakeup_req)) ||
(sleeping_on_wfi && wfi_wakeup_req);
// Note: we assert our power up request during reset, and *assume* that the
// power up acknowledge is also high at reset. If this is a problem, extend
// the core reset.
always @ (posedge clk_always_on or negedge rst_n) begin
if (!rst_n) begin
state <= S_AWAKE;
pwrup_req <= 1'b1;
clk_en <= 1'b1;
stall_release <= 1'b0;
end else begin
stall_release <= 1'b0;
case (state)
S_AWAKE: if (sleeping_on_wfi || sleeping_on_block) begin
if (stall_release) begin
// The last cycle of an ongoing which we have just released. Sit
// tight, this instruction will move down the pipeline at the
// end of this cycle. (There is an assertion that this doesn't
// happen twice.)
state <= S_AWAKE;
end else if (active_wake_req) begin
// Skip deep sleep if it would immediately fall through.
stall_release <= 1'b1;
end else if ((allow_power_down || allow_clkgate) && (sleeping_on_wfi || allow_sleep_on_block)) begin
if (frontend_pwrdown_ok) begin
pwrup_req <= !allow_power_down;
clk_en <= !allow_clkgate;
state <= allow_power_down ? S_ENTER_ASLEEP : S_ASLEEP;
end else begin
// Stay awake until it is safe to power down (i.e. until our
// instruction fetch goes quiet).
state <= S_AWAKE;
end
end else begin
// No power state change. Just sit with the pipeline stalled.
state <= S_AWAKE;
end
end
S_ENTER_ASLEEP: if (!pwrup_ack) begin
state <= S_ASLEEP;
end
S_ASLEEP: if (active_wake_req) begin
pwrup_req <= 1'b1;
clk_en <= 1'b1;
// Still go through the enter state for non-power-down wakeup, in
// case the clock gate cell has a 1 cycle delay.
state <= S_ENTER_AWAKE;
end
S_ENTER_AWAKE: if (pwrup_ack || !allow_power_down) begin
state <= S_AWAKE;
stall_release <= 1'b1;
end
default: begin
state <= S_AWAKE;
end
endcase
end
end
`ifdef HAZARD3_ASSERTIONS
// Regs are a workaround for the non-constant reset value issue with
// $past() in yosys-smtbmc.
reg past_sleeping;
reg past_stall_release;
always @ (posedge clk_always_on) begin
if (!rst_n) begin
past_sleeping <= 1'b0;
past_stall_release <= 1'b0;
end else begin
past_sleeping <= sleeping_on_wfi || sleeping_on_block;
past_stall_release <= stall_release;
// These must always be mutually exclusive.
assert(!(sleeping_on_wfi && sleeping_on_block));
if (stall_release) begin
// Presumably there was a stall which we just released
assert(past_sleeping);
// Presumably we are still in that stall
assert(sleeping_on_wfi|| sleeping_on_block);
// It takes one cycle to do a release and enter a new sleep state, so a
// double release should be impossible.
assert(!past_stall_release);
end
if (state == S_ASLEEP) begin
assert(allow_power_down || allow_clkgate);
end
end
end
`endif
// ----------------------------------------------------------------------------
// Pulse->level for block wakeup
// Unblock signal is sticky: a prior unblock with no block since will cause
// the next block to immediately fall through.
always @ (posedge clk_always_on or negedge rst_n) begin
if (!rst_n) begin
block_wakeup_req <= 1'b0;
end else begin
// Note the OR takes precedence over the AND, so we don't miss a second
// unblock that arrives at the instant we wake up.
block_wakeup_req <= (block_wakeup_req && !(
sleeping_on_block && stall_release
)) || block_wakeup_req_pulse;
end
end
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+84
View File
@@ -0,0 +1,84 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Register file
// Single write port, dual read port
`default_nettype none
module hazard3_regfile_1w2r #(
`include "hazard3_config.vh"
) (
input wire clk,
input wire rst_n,
input wire [4:0] raddr1,
output reg [W_DATA-1:0] rdata1,
input wire [4:0] raddr2,
output reg [W_DATA-1:0] rdata2,
input wire [4:0] waddr,
input wire [W_DATA-1:0] wdata,
input wire wen
);
localparam N_REGS = EXTENSION_E == 0 ? 32 : 16;
localparam [4:0] REGNUM_MASK = {~|EXTENSION_E, 4'hf};
wire [4:0] raddr1_masked = raddr1 & REGNUM_MASK;
wire [4:0] raddr2_masked = raddr2 & REGNUM_MASK;
wire [4:0] waddr_masked = waddr & REGNUM_MASK;
generate
if (RESET_REGFILE) begin: real_dualport_reset
// This will presumably always be implemented with flops
reg [W_DATA-1:0] mem [0:N_REGS-1];
integer i;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
for (i = 0; i < N_REGS; i = i + 1) begin
mem[i] <= {W_DATA{1'b0}};
end
rdata1 <= {W_DATA{1'b0}};
rdata2 <= {W_DATA{1'b0}};
end else begin
if (wen) begin
mem[waddr_masked] <= wdata;
end
rdata1 <= mem[raddr1_masked];
rdata2 <= mem[raddr2_masked];
end
end
end else begin: real_dualport_noreset
// This should be inference-compatible on FPGAs with dual-port (or 1R1W) BRAMs
`ifdef YOSYS
`ifdef FPGA_ICE40
// We do not require write-to-read bypass logic on the BRAM
(* no_rw_check *)
`endif
`endif
// Optionally force use of distributed RAM on Xilinx for better timing
`ifdef HAZARD3_REGFILE_RAM_STYLE_DISTRIBUTED
(* ram_style = "distributed" *)
`endif
reg [W_DATA-1:0] mem [0:N_REGS-1];
always @ (posedge clk) begin
if (wen) begin
mem[waddr_masked] <= wdata;
end
rdata1 <= mem[raddr1_masked];
rdata2 <= mem[raddr2_masked];
end
end
endgenerate
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+377
View File
@@ -0,0 +1,377 @@
/*****************************************************************************\
| Copyright (C) 2021-2025 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// ----------------------------------------------------------------------------
// RVFI Instrumentation
// ----------------------------------------------------------------------------
// To be included into hazard3_core.v for use with riscv-formal.
// Contains some state modelling to diagnose exactly what the core is doing,
// and report this in a way RVFI understands.
// We consider instructions to "retire" as they cross the M/W pipe register.
//
// All modelling signals prefixed with rvfm (riscv-formal monitor)
// ----------------------------------------------------------------------------
// Instruction monitor
// Diagnose whether X, M contain valid in-flight instructions, to produce
// rvfi_valid signal.
wire rvfm_x_valid = fd_cir_vld >= 2 || (fd_cir_vld >= 1 && fd_cir_raw[1:0] != 2'b11);
reg rvfm_m_valid;
reg [31:0] rvfm_m_instr;
wire rvfm_m_trap = xm_except != EXCEPT_NONE && xm_except != EXCEPT_MRET && m_trap_enter_rdy;
reg rvfi_valid_r;
reg [31:0] rvfi_insn_r;
reg rvfi_trap_r;
assign rvfi_valid = rvfi_valid_r;
assign rvfi_insn = rvfi_insn_r;
assign rvfi_trap = rvfi_trap_r;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
rvfm_m_valid <= 1'b0;
rvfi_valid_r <= 1'b0;
rvfi_trap_r <= 1'b0;
rvfi_insn_r <= 32'h0;
end else begin
if (!x_stall) begin
// X instruction squashed by any trap, as it's in the branch
// shadow.
rvfm_m_valid <= |df_cir_use && !m_trap_enter_vld;
rvfm_m_instr <= {fd_cir_raw[31:16] & {16{df_cir_use[1]}}, fd_cir_raw[15:0]};
end else if (!m_stall) begin
rvfm_m_valid <= 1'b0;
end
rvfi_valid_r <= rvfm_m_valid && !m_stall;
// Instructions which experienced fetch faults are reported as
// all-zeroes, per riscv-formal docs.
rvfi_insn_r <= rvfm_m_instr & {32{
xm_except != EXCEPT_INSTR_FAULT &&
xm_except != EXCEPT_INSTR_MISALIGN
}};
rvfi_trap_r <= rvfm_m_trap;
end
end
`ifdef HAZARD3_ASSERTIONS
always @ (posedge clk) if (rst_n) begin
// Sanity checks for above
if (d_rd != 5'h0)
assert(rvfm_x_valid);
if (xm_rd != 5'h0)
assert(rvfm_m_valid);
end
`endif
// Track whether an instruction is the first of an interrupt or exception;
// when a trap happens, a flag is installed in stage X, and once a new
// instruction arrives the flag travels alongside it down to the RVFI port.
reg rvfm_x_intr;
reg rvfm_m_intr;
reg rvfi_intr_r;
always @ (posedge clk) begin
if (!rst_n) begin
rvfm_x_intr <= 1'b0;
rvfm_m_intr <= 1'b0;
rvfi_intr_r <= 1'b0;
end else begin
rvfm_x_intr <= (rvfm_x_intr && (x_stall || d_starved || fd_cir_uop_nonfinal || df_lspair_phase_next)) ||
(m_trap_enter_vld && m_trap_enter_rdy);
if (!x_stall) begin
rvfm_m_intr <= rvfm_x_intr;
end
if (!m_stall) begin
rvfi_intr_r <= rvfm_m_intr;
end
end
end
// Hazard3 is an in-order core:
reg [63:0] rvfm_retire_ctr;
assign rvfi_order = rvfm_retire_ctr;
always @ (posedge clk or negedge rst_n)
if (!rst_n)
rvfm_retire_ctr <= 0;
else if (rvfi_valid)
rvfm_retire_ctr <= rvfm_retire_ctr + 1;
assign rvfi_intr = rvfi_intr_r && rvfi_valid;
// ----------------------------------------------------------------------------
// PC and jump monitor
reg [31:0] rvfm_xm_pc;
reg [31:0] rvfm_xm_pc_next;
// Record a jump target that was issued while stalled
reg rvfm_x_saw_f_jump;
reg [31:0] rvfm_x_saw_f_jump_target;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
rvfm_x_saw_f_jump <= 1'b0;
rvfm_x_saw_f_jump_target <= 32'd0;
end else if (f_jump_now && !(m_trap_enter_vld && m_trap_enter_rdy) && (x_stall || (fd_cir_is_uop && fd_cir_uop_nonfinal))) begin
// Record fetch address issued by instruction. Note this case is gated
// on m_trap_enter_vld && m_trap_enter_rdy (not just vld). If
// !m_trap_enter_rdy it is still possible to get a new fetch address
// in the following case:
//
// * M instr is a load/store (in dphase)
//
// * IRQ is asserted, but blocked by the load/store dphase to avoid
// trashing exception PC of a potential data-phase bus fault
//
// * X instr is a jump, which stalls due to the IRQ assertion
//
// * X instr's PC goes through to frontend during stall (and would be
// flushed if the IRQ went through) because stall cannot gate fetch
// address request to avoid AHB through-path.
//
// * IRQ's trap address is not immediately accepted by frontend due to
// address-phase stall on issuing jump instr's address
//
// * IRQ deasserts on the next cycle, so its trap address is not accepted.
rvfm_x_saw_f_jump <= 1'b1;
rvfm_x_saw_f_jump_target <= f_jump_target;
end else if (!x_stall && !(fd_cir_is_uop && fd_cir_uop_nonfinal)) begin
rvfm_x_saw_f_jump <= 1'b0;
end else if (m_trap_enter_vld && m_trap_enter_rdy) begin
// E.g. trap during uop sequence
rvfm_x_saw_f_jump <= 1'b0;
end
end
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
rvfm_xm_pc <= 0;
rvfm_xm_pc_next <= 0;
end else begin
if (!x_stall) begin
// For cm.popret and cm.popretz the PC actually changes on the
// penultimate uop; ignore this and retain the initial PC. Abuse
// knowledge that the PC update is always in the atomic section,
// and the first instruction is always interruptible.
if (!(fd_cir_is_uop && fd_cir_uop_atomic)) begin
rvfm_xm_pc <= d_pc;
end
rvfm_xm_pc_next <=
f_jump_now ? f_jump_target :
rvfm_x_saw_f_jump ? rvfm_x_saw_f_jump_target :
d_pc + (fd_cir_raw[1:0] == 2'b11 ? 32'h4 : 32'h2);
end
end
end
reg [31:0] rvfi_pc_rdata_r;
reg [31:0] rvfi_pc_wdata_r;
assign rvfi_pc_rdata = rvfi_pc_rdata_r;
assign rvfi_pc_wdata = rvfi_pc_wdata_r;
always @ (posedge clk) begin
if (!m_stall) begin
rvfi_pc_rdata_r <= rvfm_xm_pc;
rvfi_pc_wdata_r <=
m_trap_enter_vld && m_trap_enter_rdy && xm_except != EXCEPT_NONE ?
m_trap_addr : rvfm_xm_pc_next;
end
end
// ----------------------------------------------------------------------------
// Register file monitor:
// When writeback is suppressed due to trap, the previous instruction is left
// in the writeback buffer (and can be re-bypassed from there). Make sure not
// to report this as a writeback on RVFI.
reg rvfm_writeback_mask;
always @ (posedge clk) begin
if (!m_stall) begin
rvfm_writeback_mask <= m_reg_wen_if_nonzero;
end
end
assign rvfi_rd_addr = mw_rd & {5{rvfm_writeback_mask}};
assign rvfi_rd_wdata = |mw_rd && rvfm_writeback_mask ? mw_result : 32'h0;
// Do not reimplement internal bypassing logic. Danger of implementing
// it correctly here but incorrectly in core.
reg [31:0] rvfm_xm_rdata1;
reg [31:0] rvfm_xm_rdata2;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
rvfm_xm_rdata1 <= 32'h0;
rvfm_xm_rdata2 <= 32'h0;
end else if (!x_stall) begin
// rs*_bypass may have garbage on them for instructions with *no*
// register operands (due to some interesting optimisations), though
// d_rs* is still driven to 0 to disable stalling on that register
// lane. riscv-formal still likes to see zeroes from x0, so fix that
// up here. This shouldn't cover up any bugs, since a
// register-operand instruction would still *use* the garbage value.
rvfm_xm_rdata1 <= |d_rs1 ? x_rs1_bypass : 32'h0;
rvfm_xm_rdata2 <= |d_rs2 ? x_rs2_bypass : 32'h0;
end
end
reg [4:0] rvfi_rs1_addr_r;
reg [4:0] rvfi_rs2_addr_r;
reg [31:0] rvfi_rs1_rdata_r;
reg [31:0] rvfi_rs2_rdata_r;
assign rvfi_rs1_addr = rvfi_rs1_addr_r;
assign rvfi_rs2_addr = rvfi_rs2_addr_r;
assign rvfi_rs1_rdata = rvfi_rs1_rdata_r;
assign rvfi_rs2_rdata = rvfi_rs2_rdata_r;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
rvfi_rs1_addr_r <= 5'h0;
rvfi_rs2_addr_r <= 5'h0;
rvfi_rs1_rdata_r <= 32'h0;
rvfi_rs2_rdata_r <= 32'h0;
end else begin
rvfi_rs1_addr_r <= m_stall ? 5'h0 : xm_rs1;
rvfi_rs2_addr_r <= m_stall ? 5'h0 : xm_rs2;
rvfi_rs1_rdata_r <= rvfm_xm_rdata1;
rvfi_rs2_rdata_r <= xm_rs2 == mw_rd && |xm_rs2 ? m_wdata : rvfm_xm_rdata2;
end
end
// ----------------------------------------------------------------------------
// Load/store monitor: based on bus signals, NOT processor internals.
// Marshal up a description of the current data phase, and then register this
// into the RVFI signals.
`ifdef HAZARD3_ASSERTIONS
`ifndef RISCV_FORMAL_ALIGNED_MEM
initial $fatal;
`endif
`endif
reg [31:0] rvfm_haddr_dph;
reg rvfm_hwrite_dph;
reg [1:0] rvfm_htrans_dph;
reg [2:0] rvfm_hsize_dph;
always @ (posedge clk) begin
if (bus_aph_ready_d) begin
rvfm_htrans_dph <= {bus_aph_req_d, 1'b0};
rvfm_haddr_dph <= bus_haddr_d;
rvfm_hwrite_dph <= bus_hwrite_d;
rvfm_hsize_dph <= bus_hsize_d;
end
end
wire [3:0] rvfm_mem_bytemask_dph = (
rvfm_hsize_dph == 3'h0 ? 4'h1 :
rvfm_hsize_dph == 3'h1 ? 4'h3 :
4'hf
) << rvfm_haddr_dph[1:0];
reg [31:0] rvfi_mem_addr_r;
reg [3:0] rvfi_mem_rmask_r;
reg [31:0] rvfi_mem_rdata_r;
reg [3:0] rvfi_mem_wmask_r;
reg [31:0] rvfi_mem_wdata_r;
reg rvfi_mem_fault_r;
// May have to hold the strobes for multiple cycles following a bus
// fault, as the trap entry may not go through immediately (depending
// on instruction-side bus stall):
reg rvfm_mem_hold;
assign rvfi_mem_addr = rvfi_mem_addr_r;
assign rvfi_mem_rdata = rvfi_mem_rdata_r;
assign rvfi_mem_wdata = rvfi_mem_wdata_r;
assign rvfi_mem_fault = rvfi_mem_fault_r;
assign rvfi_mem_wmask = rvfi_mem_wmask_r & {4{!rvfi_mem_fault_r}};
assign rvfi_mem_rmask = rvfi_mem_rmask_r & {4{!rvfi_mem_fault_r}};
assign rvfi_mem_fault_rmask = rvfi_mem_rmask_r & {4{ rvfi_mem_fault_r}};
assign rvfi_mem_fault_wmask = rvfi_mem_wmask_r & {4{ rvfi_mem_fault_r}};
always @ (posedge clk) begin
rvfm_mem_hold <= (rvfm_mem_hold || (rvfm_htrans_dph && bus_dph_ready_d)) && m_stall;
if (xm_memop == MEMOP_AMO) begin
// AMO has completed in stage X. Progressing to stage M without MEMOP
// going to NONE then there has been no trap, therefore no stall,
// therefore no time for another address to have issued:
assert(!m_stall);
rvfi_mem_addr_r <= rvfm_haddr_dph;
// Always 32-bit, always both read and write:
rvfi_mem_rmask_r <= 4'hf;
rvfi_mem_wmask_r <= 4'hf;
// Has been juggled since the read that matched the winning write:
rvfi_mem_rdata_r <= xm_result;
// Incidentally captured on previous cycle:
rvfi_mem_wdata_r <= rvfi_mem_wdata_r;
end else if (bus_dph_ready_d) begin
// RVFI has an AXI-like concept of byte strobes, rather than AHB-like
rvfi_mem_addr_r <= rvfm_haddr_dph & 32'hffff_fffc;
{rvfi_mem_rmask_r, rvfi_mem_wmask_r} <= 0;
if (rvfm_htrans_dph[1] && rvfm_hwrite_dph) begin
rvfi_mem_wmask_r <= rvfm_mem_bytemask_dph;
rvfi_mem_wdata_r <= bus_wdata_d;
end else if (rvfm_htrans_dph[1] && !rvfm_hwrite_dph) begin
rvfi_mem_rmask_r <= rvfm_mem_bytemask_dph;
rvfi_mem_rdata_r <= bus_rdata_d;
end
rvfi_mem_fault_r <= bus_dph_err_d;
end else if (!rvfm_mem_hold) begin
rvfi_mem_rmask_r <= 4'h0;
rvfi_mem_wmask_r <= 4'h0;
rvfi_mem_fault_r <= 1'b0;
end
// Also need to report rvfi_mem_fault on a fetch fault
if (xm_except == EXCEPT_INSTR_FAULT && m_trap_enter_vld && m_trap_enter_rdy) begin
rvfi_mem_fault_r <= 1'b1;
end
end
// ----------------------------------------------------------------------------
// Constraints
// Trying to keep internal constraints to a minimum.
// Limit sleep duration for liveness checks
// TODO is it possible to do this in a way that doesn't assume the wakeup logic is functional?
`ifdef RISCV_FORMAL_FAIRNESS
reg [7:0] rvfm_sleep_counter;
always @ (posedge clk) begin
if (!rst_n) begin
rvfm_sleep_counter <= 8'd00;
end else if (xm_sleep_wfi || xm_sleep_block) begin
rvfm_sleep_counter <= rvfm_sleep_counter + 8'h01;
assume(rvfm_sleep_counter < 8'd5);
end else begin
rvfm_sleep_counter <= 8'd00;
end
end
`endif
// ----------------------------------------------------------------------------
// Tie-offs
// Note: Hazard3 does not have any instructions which irrversibly halt
// execution. For the liveness check (RISCV_FORMAL_FAIRNESS is defined),
// length of stalls is constrained and WFI is assumed to wake immediately
// after going to sleep.
assign rvfi_halt = 1'b0;
// Note: this always reports M-mode, which is not correct if the U_MODE config
// is set. However no riscv-formal checks currently use this signal.
assign rvfi_mode = 2'h3;
// Maximum XLEN is always 32 bits
assign rvfi_ixl = 2'h1;
+96
View File
@@ -0,0 +1,96 @@
/*****************************************************************************\
| Copyright (C) 2021-2025 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// Macros for building Hazard3 with RVFI trace port outside of the
// riscv-formal test harness. For example, when including the trace port in a
// synthesised processor.
`ifdef HAZARD3_RVFI_STANDALONE
`define RISCV_FORMAL
`define RISCV_FORMAL_NRET 1
`define RISCV_FORMAL_XLEN 32
`define RISCV_FORMAL_ILEN 32
`define RISCV_FORMAL_MEM_FAULT
`define RISCV_FORMAL_ALIGNED_MEM
`define RVFI_OUTPUTS \
output wire rvfi_valid, \
output wire [63:0] rvfi_order, \
output wire [31:0] rvfi_insn, \
output wire rvfi_trap, \
output wire rvfi_halt, \
output wire rvfi_intr, \
output wire [1:0] rvfi_mode, \
output wire [1:0] rvfi_ixl, \
output wire [4:0] rvfi_rs1_addr, \
output wire [4:0] rvfi_rs2_addr, \
output wire [31:0] rvfi_rs1_rdata, \
output wire [31:0] rvfi_rs2_rdata, \
output wire [4:0] rvfi_rd_addr, \
output wire [31:0] rvfi_rd_wdata, \
output wire [31:0] rvfi_pc_rdata, \
output wire [31:0] rvfi_pc_wdata, \
output wire [31:0] rvfi_mem_addr, \
output wire [3:0] rvfi_mem_rmask, \
output wire [3:0] rvfi_mem_wmask, \
output wire [31:0] rvfi_mem_rdata, \
output wire [31:0] rvfi_mem_wdata, \
output wire rvfi_mem_fault, \
output wire [3:0] rvfi_mem_fault_rmask, \
output wire [3:0] rvfi_mem_fault_wmask
`define RVFI_WIRES \
wire rvfi_valid; \
wire [63:0] rvfi_order; \
wire [31:0] rvfi_insn; \
wire rvfi_trap; \
wire rvfi_halt; \
wire rvfi_intr; \
wire [1:0] rvfi_mode; \
wire [1:0] rvfi_ixl; \
wire [4:0] rvfi_rs1_addr; \
wire [4:0] rvfi_rs2_addr; \
wire [31:0] rvfi_rs1_rdata; \
wire [31:0] rvfi_rs2_rdata; \
wire [4:0] rvfi_rd_addr; \
wire [31:0] rvfi_rd_wdata; \
wire [31:0] rvfi_pc_rdata; \
wire [31:0] rvfi_pc_wdata; \
wire [31:0] rvfi_mem_addr; \
wire [3:0] rvfi_mem_rmask; \
wire [3:0] rvfi_mem_wmask; \
wire [31:0] rvfi_mem_rdata; \
wire [31:0] rvfi_mem_wdata; \
wire rvfi_mem_fault; \
wire [3:0] rvfi_mem_fault_rmask; \
wire [3:0] rvfi_mem_fault_wmask;
`define RVFI_CONN \
.rvfi_valid (rvfi_valid), \
.rvfi_order (rvfi_order), \
.rvfi_insn (rvfi_insn), \
.rvfi_trap (rvfi_trap), \
.rvfi_halt (rvfi_halt), \
.rvfi_intr (rvfi_intr), \
.rvfi_mode (rvfi_mode), \
.rvfi_ixl (rvfi_ixl), \
.rvfi_rs1_addr (rvfi_rs1_addr), \
.rvfi_rs2_addr (rvfi_rs2_addr), \
.rvfi_rs1_rdata (rvfi_rs1_rdata), \
.rvfi_rs2_rdata (rvfi_rs2_rdata), \
.rvfi_rd_addr (rvfi_rd_addr), \
.rvfi_rd_wdata (rvfi_rd_wdata), \
.rvfi_pc_rdata (rvfi_pc_rdata), \
.rvfi_pc_wdata (rvfi_pc_wdata), \
.rvfi_mem_addr (rvfi_mem_addr), \
.rvfi_mem_rmask (rvfi_mem_rmask), \
.rvfi_mem_wmask (rvfi_mem_wmask), \
.rvfi_mem_rdata (rvfi_mem_rdata), \
.rvfi_mem_wdata (rvfi_mem_wdata), \
.rvfi_mem_fault (rvfi_mem_fault), \
.rvfi_mem_fault_rmask (rvfi_mem_fault_rmask), \
.rvfi_mem_fault_wmask (rvfi_mem_fault_wmask)
`endif
+499
View File
@@ -0,0 +1,499 @@
/*****************************************************************************\
| Copyright (C) 2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
`default_nettype none
// The Hazard3 trigger unit always implements one trigger of each of the
// following types:
//
// * Instruction count trigger (type=3) with count=1 (can single-step U-mode
// from M-mode, or step M-mode foreground from an M-mode exception handler)
//
// * Interrupt trigger (type=4): trigger on mask of mtip/msip/meip interrupts
//
// * Exception trigger (type=5): trigger on mask of exception causes
//
// The following are optionally supported:
//
// * Instruction address triggers (type=2 execute=1 select=0) aka breakpoints
//
// Breakpoints always use exact address matches, and the timing is always
// "early". The number of breakpoints is configured by BREAKPOINT_TRIGGERS,
// which can be 0.
//
// Interrupt/exception triggers break after the core transfers to its trap
// handler, but before the first trap handler instruction executes. Only
// action=1 is supported for these triggers, since trap-on-trap is useless
// when M is the only privileged mode.
module hazard3_triggers #(
`include "hazard3_config.vh"
) (
input wire clk,
input wire rst_n,
// Config interface passed through CSR block
input wire [11:0] cfg_addr,
input wire cfg_wen,
input wire [W_DATA-1:0] cfg_wdata,
output reg [W_DATA-1:0] cfg_rdata,
// Global trigger-to-M-mode enable from tcontrol
input wire trig_m_en,
// Fetch address query from stage F
input wire [W_ADDR-1:0] fetch_addr,
input wire fetch_m_mode,
input wire fetch_d_mode,
// Trap trigger events from stage X
input wire event_instr_ret,
// Trap trigger events from stage M
input wire event_interrupt,
input wire event_exception,
input wire [3:0] event_trap_cause,
input wire event_trap_enter,
// F-aligned break request (for each halfword of word-sized word-aligned fetch)
output wire [1:0] break_any,
output wire [1:0] break_d_mode,
// X-aligned step break request (to M-mode only)
output wire break_m_step,
// Stage-X debug mode flag, for CSR protection (may or may not be the same
// as the query debug mode flag)
input wire x_d_mode,
// Stage-X M-mode flag, for enables on interrupt/exception triggers
input wire x_m_mode
);
`include "hazard3_csr_addr.vh"
generate
if (DEBUG_SUPPORT == 0) begin: no_triggers
// The instantiation of this block should already be stubbed out in core.v if
// there are no triggers, but we still get warnings for elaborating this
// module with zero triggers, so add a generate block here too.
always @ (*) cfg_rdata = {W_DATA{1'b0}};
assign break_any = 1'b0;
assign break_d_mode = 1'b0;
end else begin: have_triggers
localparam TINDEX_ICOUNT = BREAKPOINT_TRIGGERS + 0;
localparam TINDEX_INTERRUPT = BREAKPOINT_TRIGGERS + 1;
localparam TINDEX_EXCEPTION = BREAKPOINT_TRIGGERS + 2;
localparam N_TRIGGERS = BREAKPOINT_TRIGGERS + 3;
// If there are no breakpoints, we still have one dummy register (hardwired to
// zero) for Verilog wrangling purposes. It has no synthesis effect.
localparam N_BREAKPOINT_REGS = BREAKPOINT_TRIGGERS > 0 ? BREAKPOINT_TRIGGERS : 1;
// ----------------------------------------------------------------------------
// Configuration state
localparam W_TSELECT = $clog2(N_TRIGGERS);
reg [W_TSELECT-1:0] tselect;
// Note tdata1 and mcontrol are the same CSR. tdata1 refers to the universal
// fields (type/dmode) and mcontrol refers to those fields specific to
// type=2 (address/data match), the only trigger type we implement.
// State for instruction address match triggers (breakpoints).
reg bp_tdata1_dmode [0:N_BREAKPOINT_REGS-1];
reg mcontrol_action [0:N_BREAKPOINT_REGS-1];
reg mcontrol_m [0:N_BREAKPOINT_REGS-1];
reg mcontrol_u [0:N_BREAKPOINT_REGS-1];
reg mcontrol_execute [0:N_BREAKPOINT_REGS-1];
reg [W_DATA-1:0] bp_tdata2 [0:N_BREAKPOINT_REGS-1];
// State for instruction count trigger
// (hardwired: count=1 dmode=0 action=0; Debug mode single step is already
// available via dcsr)
reg icount_m;
reg icount_u;
// State for interrupt trigger
// (hardwired: action=1; M-mode trap-on-trap is useless as you lose the
// original trap state)
reg trigger_irq_m;
reg trigger_irq_u;
reg trigger_irq_dmode;
reg [15:0] trigger_irq_cause;
localparam [15:0] IMPLEMENTED_IRQ_CAUSES = {
4'h0, // reserved
1'b1, // meip
3'h0, // reserved or unimplemented
1'b1, // mtip
3'h0, // reserved or unimplemented
1'b1, // msip
3'h0 // reserved or unimplemented
};
// State for exception trigger
// (hardwired: action=1; M-mode trap-on-trap is useless as you lose the
// original trap state)
reg trigger_exception_m;
reg trigger_exception_u;
reg trigger_exception_dmode;
reg [15:0] trigger_exception_cause;
localparam [15:0] IMPLEMENTED_EXCEPTION_CAUSES = {
4'h0, // reserved
1'b1, // 11 -> ecall from M-mode
2'h0, // reserved or unimplemented
|U_MODE, // 8 -> ecall from U-mode
1'b1, // 7 -> store/AMO fault
1'b1, // 6 -> store/AMO align
1'b1, // 5 -> load fault
1'b1, // 4 -> load align
1'b0, // 3 -> breakpoint; seems useless and risky so disallow
1'b1, // 2 -> illegal opcode
1'b1, // 1 -> fetch fault
~|EXTENSION_C // 0 -> fetch align (only when IALIGN is 32-bit)
};
// ----------------------------------------------------------------------------
// Configuration write port
localparam N_TRIGGERS_PADDED = 1 << $clog2(N_TRIGGERS);
wire [N_TRIGGERS_PADDED-1:0] tselect_match = {{N_TRIGGERS_PADDED-1{1'b0}}, 1'b1} << tselect;
always @ (posedge clk or negedge rst_n) begin: cfg_update
integer i;
if (!rst_n) begin
tselect <= {W_TSELECT{1'b0}};
icount_m <= 1'b0;
icount_u <= 1'b0;
trigger_irq_m <= 1'b0;
trigger_irq_u <= 1'b0;
trigger_irq_dmode <= 1'b0;
trigger_irq_cause <= 16'h0;
trigger_exception_m <= 1'b0;
trigger_exception_u <= 1'b0;
trigger_exception_dmode <= 1'b0;
trigger_exception_cause <= 16'h0;
for (i = 0; i < BREAKPOINT_TRIGGERS; i = i + 1) begin
bp_tdata1_dmode[i] <= 1'b0;
mcontrol_action[i] <= 1'b0;
mcontrol_m[i] <= 1'b0;
mcontrol_u[i] <= 1'b0;
mcontrol_execute[i] <= 1'b0;
bp_tdata2[i] <= {W_DATA{1'b0}};
end
end else begin
if (cfg_wen && cfg_addr == TSELECT) begin
tselect <= cfg_wdata[W_TSELECT-1:0];
end else if (cfg_wen && cfg_addr == TDATA1) begin
if (tselect_match[TINDEX_ICOUNT]) begin
// This trigger does not implement a dmode bit, as Debug-mode
// break on single-step is already provided by dcsr.step
icount_m <= cfg_wdata[9];
icount_u <= cfg_wdata[6] && |U_MODE;
end
if (tselect_match[TINDEX_INTERRUPT] && !(trigger_irq_dmode && !x_d_mode)) begin
trigger_irq_dmode <= cfg_wdata[27];
trigger_irq_m <= cfg_wdata[9];
trigger_irq_u <= cfg_wdata[6] && |U_MODE;
end
if (tselect_match[TINDEX_EXCEPTION] && !(trigger_exception_dmode && !x_d_mode)) begin
trigger_exception_dmode <= cfg_wdata[27];
trigger_exception_m <= cfg_wdata[9];
trigger_exception_u <= cfg_wdata[6] && |U_MODE;
end
for (i = 0; i < BREAKPOINT_TRIGGERS; i = i + 1) begin
if (tselect_match[i] && !(bp_tdata1_dmode[i] && !x_d_mode)) begin
if (x_d_mode) begin
bp_tdata1_dmode[i] <= cfg_wdata[27];
end
mcontrol_action[i] <= cfg_wdata[12];
mcontrol_m[i] <= cfg_wdata[6];
mcontrol_u[i] <= cfg_wdata[3] && |U_MODE;
mcontrol_execute[i] <= cfg_wdata[2];
end
end
end else if (cfg_wen && cfg_addr == TDATA2) begin
if (tselect_match[TINDEX_INTERRUPT] && !(trigger_irq_dmode && !x_d_mode)) begin
trigger_irq_cause <= cfg_wdata[15:0] & IMPLEMENTED_IRQ_CAUSES;
end
if (tselect_match[TINDEX_EXCEPTION] && !(trigger_exception_dmode && !x_d_mode)) begin
trigger_exception_cause <= cfg_wdata[15:0] & IMPLEMENTED_EXCEPTION_CAUSES;
end
for (i = 0; i < BREAKPOINT_TRIGGERS; i = i + 1) begin
if (tselect_match[i] && !(bp_tdata1_dmode[i] && !x_d_mode)) begin
bp_tdata2[i] <= cfg_wdata & {{W_ADDR-2{1'b1}}, |EXTENSION_C, 1'b0};
end
end
end
if ((x_d_mode || event_trap_enter) && break_m_step) begin
// count field is hardwired, so the trigger is required to disable
// itself by clearing its own enables
icount_m <= 1'b0;
icount_u <= 1'b0;
end
// With no breakpoints, there is still a dummy entry to avoid
// `generate` spaghetti; tools complain about comb processes without
// sensitivities etc, so just synchronously tie to 0:
if (BREAKPOINT_TRIGGERS == 0) begin
bp_tdata1_dmode[0] <= 1'b0;
mcontrol_action[0] <= 1'b0;
mcontrol_m[0] <= 1'b0;
mcontrol_u[0] <= 1'b0;
mcontrol_execute[0] <= 1'b0;
bp_tdata2[0] <= {W_DATA{1'b0}};
end
end
end
// ----------------------------------------------------------------------------
// Configuration read port
reg [W_DATA-1:0] tdata1_rdata [0:N_TRIGGERS_PADDED-1];
reg [W_DATA-1:0] tdata2_rdata [0:N_TRIGGERS_PADDED-1];
reg [W_DATA-1:0] tinfo_rdata [0:N_TRIGGERS_PADDED-1];
always @ (*) begin: generate_padded_rdata
// Default for unimplemented triggers
integer i;
for (i = 0; i < N_TRIGGERS_PADDED; i = i + 1) begin
tdata1_rdata[i] = {W_DATA{1'b0}};
tdata2_rdata[i] = {W_DATA{1'b0}};
tinfo_rdata[i] = 32'd1 << 0; // type = 0, no trigger
end
// Breakpoints are the first n triggers
for (i = 0; i < BREAKPOINT_TRIGGERS; i = i + 1) begin
tdata1_rdata[i] = {
4'h2, // type = address/data match
bp_tdata1_dmode[i],
6'h00, // maskmax = 0, exact match only
1'b0, // hit = 0, not implemented
1'b0, // select = 0, address match only
1'b0, // timing = 0, trigger before execution
2'h0, // sizelo = 0, unsized
{3'h0, mcontrol_action[i]}, // action = 0/1, break to M-mode/D-mode
1'b0, // chain = 0, chaining is useless for exact matches
4'h0, // match = 0, exact match only
mcontrol_m[i],
1'b0,
1'b0, // s = 0, no S-mode
mcontrol_u[i],
mcontrol_execute[i],
1'b0, // store = 0, this is not a watchpoint
1'b0 // load = 0, this is not a watchpoint
};
tdata2_rdata[i] = bp_tdata2[i];
tinfo_rdata[i] = 32'd1 << 2; // type = 2, address/data match
end
// Instruction count trigger
tdata1_rdata[TINDEX_ICOUNT] = {
4'h3, // type = instruction count
1'b0, // dmode = 0 (Debug mode already has dcsr.step)
2'h0, // reserved
1'b0, // hit = 0
14'd1, // count = 1, single-step only
icount_m,
1'b0, // reserved
1'b0, // s = 0, no S-mode
icount_u,
6'h0 // action = 0, break to M-mode
};
tinfo_rdata[TINDEX_ICOUNT] = 32'd1 << 3; // type = 3, instruction count
// Interrupt trigger
tdata1_rdata[TINDEX_INTERRUPT] = {
4'h4, // type = interrupt
trigger_irq_dmode,
1'b0, // hit = 0
16'h0, // reserved
trigger_irq_m,
1'b0, // reserved
1'b0, // s = 0, no S-mode
trigger_irq_u,
6'd1 // action = 1, break to Debug mode (if dmode=1)
};
tdata2_rdata[TINDEX_INTERRUPT] = {
16'h0,
trigger_irq_cause & IMPLEMENTED_IRQ_CAUSES
};
tinfo_rdata[TINDEX_INTERRUPT] = 32'd1 << 4;
// Exception trigger
tdata1_rdata[TINDEX_EXCEPTION] = {
4'h5, // type = exception
trigger_exception_dmode,
1'b0, // hit = 0
16'h0, // reserved
trigger_exception_m,
1'b0, // reserved
1'b0, // s = 0, no S-mode
trigger_exception_u,
6'd1 // action = 1, break to Debug mode (if dmode=1)
};
tdata2_rdata[TINDEX_EXCEPTION] = {
16'h0,
trigger_exception_cause & IMPLEMENTED_EXCEPTION_CAUSES
};
tinfo_rdata[TINDEX_EXCEPTION] = 32'd1 << 5;
end
always @ (*) begin
cfg_rdata = {W_DATA{1'b0}};
if (cfg_addr == TSELECT) begin
cfg_rdata = {{W_DATA-W_TSELECT{1'b0}}, tselect};
end else if (cfg_addr == TDATA1) begin
cfg_rdata = tdata1_rdata[tselect];
end else if (cfg_addr == TDATA2) begin
cfg_rdata = tdata2_rdata[tselect];
end else if (cfg_addr == TINFO) begin
cfg_rdata = tinfo_rdata[tselect];
end
end
// ----------------------------------------------------------------------------
// Interrupt/exception trigger logic
// Ignore tcontrol.mte as these triggers never target M-mode.
wire exception_trigger_match =
!x_d_mode && trigger_exception_dmode &&
(x_m_mode ? trigger_exception_m : trigger_exception_u) &&
event_exception &&
trigger_exception_cause[event_trap_cause] &&
IMPLEMENTED_EXCEPTION_CAUSES[event_trap_cause];
wire interrupt_trigger_match =
!x_d_mode && trigger_irq_dmode &&
(x_m_mode ? trigger_irq_m : trigger_irq_u) &&
event_interrupt &&
trigger_irq_cause[event_trap_cause] &&
IMPLEMENTED_IRQ_CAUSES[event_trap_cause];
// Asserted no later than the end of the aphase for the instruction fetch at
// mtvec. Tags the dphase of trap handler instruction fetches as containing
// breakpoints.
reg break_ie;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
break_ie <= 1'b0;
end else begin
break_ie <= !x_d_mode && (break_ie || (
exception_trigger_match || interrupt_trigger_match
));
end
end
// ----------------------------------------------------------------------------
// Instruction count trigger logic (single-step under M-mode control)
wire step_break_enabled = trig_m_en && !x_d_mode && (
x_m_mode ? icount_m : icount_u
);
reg break_on_step;
always @ (posedge clk or negedge rst_n) begin
if (!rst_n) begin
break_on_step <= 1'b0;
end else begin
// Note icount triggers differ from dcsr.step in that they ignore
// exceptions, only triggering on retired instructions.
break_on_step <= !(x_d_mode || event_trap_enter) && (break_on_step || (
event_instr_ret && step_break_enabled
));
end
end
assign break_m_step = break_on_step;
// ----------------------------------------------------------------------------
// Breakpoint trigger logic
// To reduce the fanin of jump and load/store gating in stage X, the address
// lookup is in stage F (fetch data phase). We check *fetch addresses*, not
// program counter values. Fetches are always word-sized and word-aligned.
//
// To ensure it is safe to do this, non-debug-mode writes to the TDATA1 and
// TDATA2 CSRs cause a prefetch flush, to maintain write-to-fetch ordering.
//
// It's possible for different breakpoints to match different halfwords of the
// fetch word. The trigger unit must report both matches separately, because
// it is not known at this point where the instruction boundaries are (we
// don't have the instruction data yet).
wire [N_BREAKPOINT_REGS-1:0] breakpoint_enabled;
wire [N_BREAKPOINT_REGS-1:0] breakpoint_match;
wire [N_BREAKPOINT_REGS-1:0] want_d_mode_break;
wire [N_BREAKPOINT_REGS-1:0] want_m_mode_break;
wire [N_BREAKPOINT_REGS-1:0] want_d_mode_break_hw0;
wire [N_BREAKPOINT_REGS-1:0] want_d_mode_break_hw1;
wire [N_BREAKPOINT_REGS-1:0] want_m_mode_break_hw0;
wire [N_BREAKPOINT_REGS-1:0] want_m_mode_break_hw1;
genvar g;
for (g = 0; g < N_BREAKPOINT_REGS; g = g + 1) begin: match_pc
// Detect tripped breakpoints
assign breakpoint_enabled[g] = mcontrol_execute[g] && !fetch_d_mode && (
fetch_m_mode ? mcontrol_m[g] : mcontrol_u[g]
);
assign breakpoint_match[g] = breakpoint_enabled[g] && fetch_addr == {bp_tdata2[g][W_DATA-1:2], 2'b00};
// Decide the type of break implied by the trip
assign want_d_mode_break[g] = breakpoint_match[g] && mcontrol_action[g] && bp_tdata1_dmode[g];
assign want_m_mode_break[g] = breakpoint_match[g] && !mcontrol_action[g] && trig_m_en;
// Report separately for each halfword, so the frontend can pass this
// through the prefetch buffer. A breakpoint exception is taken when
// the first halfword of an instruction (of any size) is flagged with
// a breakpoint, implying an exact match.
assign want_d_mode_break_hw0[g] = want_d_mode_break[g] && !bp_tdata2[g][1];
assign want_d_mode_break_hw1[g] = want_d_mode_break[g] && bp_tdata2[g][1];
assign want_m_mode_break_hw0[g] = want_m_mode_break[g] && !bp_tdata2[g][1];
assign want_m_mode_break_hw1[g] = want_m_mode_break[g] && bp_tdata2[g][1];
end
// Break flags to frontend (tag the current fetch dphase as containing a breakpoint):
assign break_any = {
|want_m_mode_break_hw1 || |want_d_mode_break_hw1 || break_ie,
|want_m_mode_break_hw0 || |want_d_mode_break_hw0 || break_ie
} & {2{BREAKPOINT_TRIGGERS > 0}};
assign break_d_mode = {
|want_d_mode_break_hw1 || break_ie,
|want_d_mode_break_hw0 || break_ie
} & {2{BREAKPOINT_TRIGGERS > 0}};
end
endgenerate
endmodule
`ifndef YOSYS
`default_nettype wire
`endif
+20
View File
@@ -0,0 +1,20 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
// These really ought to be localparams, but are occasionally needed for
// passing flags around between modules, so are made available as parameters
// instead. It's ugly, but better scope hygiene than the preprocessor. These
// parameters should not be changed from their default values.
parameter W_REGADDR = 5,
parameter W_ALUOP = 6,
parameter W_ALUSRC = 1,
parameter W_MEMOP = 5,
parameter W_BCOND = 2,
parameter W_SHAMT = 5,
parameter W_EXCEPT = 4,
parameter W_MULOP = 3
+276
View File
@@ -0,0 +1,276 @@
/*****************************************************************************\
| Copyright (C) 2021-2022 Luke Wren |
| SPDX-License-Identifier: Apache-2.0 |
\*****************************************************************************/
localparam RV_RS1_LSB = 15;
localparam RV_RS1_BITS = 5;
localparam RV_RS2_LSB = 20;
localparam RV_RS2_BITS = 5;
localparam RV_RD_LSB = 7;
localparam RV_RD_BITS = 5;
// Note: these are preprocessor macros, rather than the usual localparams,
// because it's quite difficult to get a definitive citation from 1364-2005
// for whether Z values are propagated through a localparam to a casez.
// Multiple tools complain about it, so just this once I'll use macros.
`ifndef HAZARD3_RVOPC_MACROS
`define HAZARD3_RVOPC_MACROS
// Base ISA (some of these are Z now)
`define RVOPC_BEQ 32'b?????????????????000?????1100011
`define RVOPC_BNE 32'b?????????????????001?????1100011
`define RVOPC_BLT 32'b?????????????????100?????1100011
`define RVOPC_BGE 32'b?????????????????101?????1100011
`define RVOPC_BLTU 32'b?????????????????110?????1100011
`define RVOPC_BGEU 32'b?????????????????111?????1100011
`define RVOPC_JALR 32'b?????????????????000?????1100111
`define RVOPC_JAL 32'b?????????????????????????1101111
`define RVOPC_LUI 32'b?????????????????????????0110111
`define RVOPC_AUIPC 32'b?????????????????????????0010111
`define RVOPC_ADDI 32'b?????????????????000?????0010011
`define RVOPC_SLLI 32'b0000000??????????001?????0010011
`define RVOPC_SLTI 32'b?????????????????010?????0010011
`define RVOPC_SLTIU 32'b?????????????????011?????0010011
`define RVOPC_XORI 32'b?????????????????100?????0010011
`define RVOPC_SRLI 32'b0000000??????????101?????0010011
`define RVOPC_SRAI 32'b0100000??????????101?????0010011
`define RVOPC_ORI 32'b?????????????????110?????0010011
`define RVOPC_ANDI 32'b?????????????????111?????0010011
`define RVOPC_ADD 32'b0000000??????????000?????0110011
`define RVOPC_SUB 32'b0100000??????????000?????0110011
`define RVOPC_SLL 32'b0000000??????????001?????0110011
`define RVOPC_SLT 32'b0000000??????????010?????0110011
`define RVOPC_SLTU 32'b0000000??????????011?????0110011
`define RVOPC_XOR 32'b0000000??????????100?????0110011
`define RVOPC_SRL 32'b0000000??????????101?????0110011
`define RVOPC_SRA 32'b0100000??????????101?????0110011
`define RVOPC_OR 32'b0000000??????????110?????0110011
`define RVOPC_AND 32'b0000000??????????111?????0110011
`define RVOPC_LB 32'b?????????????????000?????0000011
`define RVOPC_LH 32'b?????????????????001?????0000011
`define RVOPC_LW 32'b?????????????????010?????0000011
`define RVOPC_LBU 32'b?????????????????100?????0000011
`define RVOPC_LHU 32'b?????????????????101?????0000011
`define RVOPC_SB 32'b?????????????????000?????0100011
`define RVOPC_SH 32'b?????????????????001?????0100011
`define RVOPC_SW 32'b?????????????????010?????0100011
`define RVOPC_FENCE 32'b????????????00000000000000001111
`define RVOPC_FENCE_I 32'b00000000000000000001000000001111
`define RVOPC_ECALL 32'b00000000000000000000000001110011
`define RVOPC_EBREAK 32'b00000000000100000000000001110011
`define RVOPC_CSRRW 32'b?????????????????001?????1110011
`define RVOPC_CSRRS 32'b?????????????????010?????1110011
`define RVOPC_CSRRC 32'b?????????????????011?????1110011
`define RVOPC_CSRRWI 32'b?????????????????101?????1110011
`define RVOPC_CSRRSI 32'b?????????????????110?????1110011
`define RVOPC_CSRRCI 32'b?????????????????111?????1110011
`define RVOPC_MRET 32'b00110000001000000000000001110011
`define RVOPC_SYSTEM 32'b?????????????????????????1110011
`define RVOPC_WFI 32'b00010000010100000000000001110011
// M extension
`define RVOPC_MUL 32'b0000001??????????000?????0110011
`define RVOPC_MULH 32'b0000001??????????001?????0110011
`define RVOPC_MULHSU 32'b0000001??????????010?????0110011
`define RVOPC_MULHU 32'b0000001??????????011?????0110011
`define RVOPC_DIV 32'b0000001??????????100?????0110011
`define RVOPC_DIVU 32'b0000001??????????101?????0110011
`define RVOPC_REM 32'b0000001??????????110?????0110011
`define RVOPC_REMU 32'b0000001??????????111?????0110011
// A extension
`define RVOPC_LR_W 32'b00010??00000?????010?????0101111
`define RVOPC_SC_W 32'b00011????????????010?????0101111
`define RVOPC_AMOSWAP_W 32'b00001????????????010?????0101111
`define RVOPC_AMOADD_W 32'b00000????????????010?????0101111
`define RVOPC_AMOXOR_W 32'b00100????????????010?????0101111
`define RVOPC_AMOAND_W 32'b01100????????????010?????0101111
`define RVOPC_AMOOR_W 32'b01000????????????010?????0101111
`define RVOPC_AMOMIN_W 32'b10000????????????010?????0101111
`define RVOPC_AMOMAX_W 32'b10100????????????010?????0101111
`define RVOPC_AMOMINU_W 32'b11000????????????010?????0101111
`define RVOPC_AMOMAXU_W 32'b11100????????????010?????0101111
// Zba (address generation)
`define RVOPC_SH1ADD 32'b0010000??????????010?????0110011
`define RVOPC_SH2ADD 32'b0010000??????????100?????0110011
`define RVOPC_SH3ADD 32'b0010000??????????110?????0110011
// Zbb (basic bit manipulation)
`define RVOPC_ANDN 32'b0100000??????????111?????0110011
`define RVOPC_CLZ 32'b011000000000?????001?????0010011
`define RVOPC_CPOP 32'b011000000010?????001?????0010011
`define RVOPC_CTZ 32'b011000000001?????001?????0010011
`define RVOPC_MAX 32'b0000101??????????110?????0110011
`define RVOPC_MAXU 32'b0000101??????????111?????0110011
`define RVOPC_MIN 32'b0000101??????????100?????0110011
`define RVOPC_MINU 32'b0000101??????????101?????0110011
`define RVOPC_ORC_B 32'b001010000111?????101?????0010011
`define RVOPC_ORN 32'b0100000??????????110?????0110011
`define RVOPC_REV8 32'b011010011000?????101?????0010011
`define RVOPC_ROL 32'b0110000??????????001?????0110011
`define RVOPC_ROR 32'b0110000??????????101?????0110011
`define RVOPC_RORI 32'b0110000??????????101?????0010011
`define RVOPC_SEXT_B 32'b011000000100?????001?????0010011
`define RVOPC_SEXT_H 32'b011000000101?????001?????0010011
`define RVOPC_XNOR 32'b0100000??????????100?????0110011
`define RVOPC_ZEXT_H 32'b000010000000?????100?????0110011
// Zbc (carry-less multiply)
`define RVOPC_CLMUL 32'b0000101??????????001?????0110011
`define RVOPC_CLMULH 32'b0000101??????????011?????0110011
`define RVOPC_CLMULR 32'b0000101??????????010?????0110011
// Zbs (single-bit manipulation)
`define RVOPC_BCLR 32'b0100100??????????001?????0110011
`define RVOPC_BCLRI 32'b0100100??????????001?????0010011
`define RVOPC_BEXT 32'b0100100??????????101?????0110011
`define RVOPC_BEXTI 32'b0100100??????????101?????0010011
`define RVOPC_BINV 32'b0110100??????????001?????0110011
`define RVOPC_BINVI 32'b0110100??????????001?????0010011
`define RVOPC_BSET 32'b0010100??????????001?????0110011
`define RVOPC_BSETI 32'b0010100??????????001?????0010011
// Zbkb (basic bit manipulation for crypto) (minus those in Zbb)
`define RVOPC_PACK 32'b0000100??????????100?????0110011
`define RVOPC_PACKH 32'b0000100??????????111?????0110011
`define RVOPC_BREV8 32'b011010000111?????101?????0010011
`define RVOPC_UNZIP 32'b000010001111?????101?????0010011
`define RVOPC_ZIP 32'b000010001111?????001?????0010011
// Zbkc is a subset of Zbc.
// Zbkx (crossbar permutation)
`define RVOPC_XPERM8 32'b0010100??????????100?????0110011
`define RVOPC_XPERM4 32'b0010100??????????010?????0110011
// Zilsd (load/store pair)
`define RVOPC_LD 32'b?????????????????011????00000011 // rd[0] == 0
`define RVOPC_SD 32'b???????????0?????011?????0100011 // rs2[0] == 0
// Hazard3 custom instructions
// Xh3bextm (Hazard3 multi-bit extract): multi-bit versions of bext/bexti from Zbs
`define RVOPC_H3_BEXTM 32'b000???0??????????000?????0001011 // custom-0 funct3=0
`define RVOPC_H3_BEXTMI 32'b000???0??????????100?????0001011 // custom-0 funct3=4
// C Extension
`define RVOPC_C_ADDI4SPN 16'b000???????????00 // *** illegal if imm 0
`define RVOPC_C_LW 16'b010???????????00
`define RVOPC_C_SW 16'b110???????????00
`define RVOPC_C_ADDI 16'b000???????????01
`define RVOPC_C_JAL 16'b001???????????01
`define RVOPC_C_J 16'b101???????????01
`define RVOPC_C_LI 16'b010???????????01
// addi16sp when rd=2:
`define RVOPC_C_LUI 16'b011???????????01 // *** reserved if imm 0 (for both LUI and ADDI16SP)
`define RVOPC_C_SRLI 16'b100000????????01 // On RV32 imm[5] (instr[12]) must be 0, else reserved NSE.
`define RVOPC_C_SRAI 16'b100001????????01 // On RV32 imm[5] (instr[12]) must be 0, else reserved NSE.
`define RVOPC_C_ANDI 16'b100?10????????01
`define RVOPC_C_SUB 16'b100011???00???01
`define RVOPC_C_XOR 16'b100011???01???01
`define RVOPC_C_OR 16'b100011???10???01
`define RVOPC_C_AND 16'b100011???11???01
`define RVOPC_C_BEQZ 16'b110???????????01
`define RVOPC_C_BNEZ 16'b111???????????01
`define RVOPC_C_SLLI 16'b0000??????????10 // On RV32 imm[5] (instr[12]) must be 0, else reserved NSE.
// jr if !rs2:
`define RVOPC_C_MV 16'b1000??????????10 // *** reserved if JR and !rs1 (instr[11:7])
// jalr if !rs2:
`define RVOPC_C_ADD 16'b1001??????????10 // *** EBREAK if !instr[11:2]
`define RVOPC_C_LWSP 16'b010???????????10 // *** reserved if rd=x0
`define RVOPC_C_SWSP 16'b110???????????10
// Zcb simple additional compressed instructions
`define RVOPC_C_LBU 16'b100000????????00
`define RVOPC_C_LHU 16'b100001???0????00
`define RVOPC_C_LH 16'b100001???1????00
`define RVOPC_C_SB 16'b100010????????00
`define RVOPC_C_SH 16'b100011???0????00
`define RVOPC_C_ZEXT_B 16'b100111???1100001
`define RVOPC_C_SEXT_B 16'b100111???1100101
`define RVOPC_C_ZEXT_H 16'b100111???1101001
`define RVOPC_C_SEXT_H 16'b100111???1101101
`define RVOPC_C_NOT 16'b100111???1110101
`define RVOPC_C_MUL 16'b100111???10???01
// Zclsd load/store pair instructions
`define RVOPC_C_LD 16'b011??????????000 // rd[0] == 0
`define RVOPC_C_LDSP 16'b011?????0?????10 // rd[0] == 0
`define RVOPC_C_SD 16'b111??????????000 // rs2[0] == 0
`define RVOPC_C_SDSP 16'b111??????????010 // rs2[0] == 0
// Zcmp push/pop instructions
`define RVOPC_CM_PUSH 16'b10111000??????10
`define RVOPC_CM_POP 16'b10111010??????10
`define RVOPC_CM_POPRETZ 16'b10111100??????10
`define RVOPC_CM_POPRET 16'b10111110??????10
`define RVOPC_CM_MVSA01 16'b101011???01???10
`define RVOPC_CM_MVA01S 16'b101011???11???10
// Copies provided here with 0 instead of ? so that these can be used to build 32-bit instructions in the decompressor
`define RVOPC_NOZ_BEQ 32'b00000000000000000000000001100011
`define RVOPC_NOZ_BNE 32'b00000000000000000001000001100011
`define RVOPC_NOZ_BLT 32'b00000000000000000100000001100011
`define RVOPC_NOZ_BGE 32'b00000000000000000101000001100011
`define RVOPC_NOZ_BLTU 32'b00000000000000000110000001100011
`define RVOPC_NOZ_BGEU 32'b00000000000000000111000001100011
`define RVOPC_NOZ_JALR 32'b00000000000000000000000001100111
`define RVOPC_NOZ_JAL 32'b00000000000000000000000001101111
`define RVOPC_NOZ_LUI 32'b00000000000000000000000000110111
`define RVOPC_NOZ_AUIPC 32'b00000000000000000000000000010111
`define RVOPC_NOZ_ADDI 32'b00000000000000000000000000010011
`define RVOPC_NOZ_SLLI 32'b00000000000000000001000000010011
`define RVOPC_NOZ_SLTI 32'b00000000000000000010000000010011
`define RVOPC_NOZ_SLTIU 32'b00000000000000000011000000010011
`define RVOPC_NOZ_XORI 32'b00000000000000000100000000010011
`define RVOPC_NOZ_SRLI 32'b00000000000000000101000000010011
`define RVOPC_NOZ_SRAI 32'b01000000000000000101000000010011
`define RVOPC_NOZ_ORI 32'b00000000000000000110000000010011
`define RVOPC_NOZ_ANDI 32'b00000000000000000111000000010011
`define RVOPC_NOZ_ADD 32'b00000000000000000000000000110011
`define RVOPC_NOZ_SUB 32'b01000000000000000000000000110011
`define RVOPC_NOZ_SLL 32'b00000000000000000001000000110011
`define RVOPC_NOZ_SLT 32'b00000000000000000010000000110011
`define RVOPC_NOZ_SLTU 32'b00000000000000000011000000110011
`define RVOPC_NOZ_XOR 32'b00000000000000000100000000110011
`define RVOPC_NOZ_SRL 32'b00000000000000000101000000110011
`define RVOPC_NOZ_SRA 32'b01000000000000000101000000110011
`define RVOPC_NOZ_OR 32'b00000000000000000110000000110011
`define RVOPC_NOZ_AND 32'b00000000000000000111000000110011
`define RVOPC_NOZ_LB 32'b00000000000000000000000000000011
`define RVOPC_NOZ_LH 32'b00000000000000000001000000000011
`define RVOPC_NOZ_LW 32'b00000000000000000010000000000011
`define RVOPC_NOZ_LBU 32'b00000000000000000100000000000011
`define RVOPC_NOZ_LHU 32'b00000000000000000101000000000011
`define RVOPC_NOZ_SB 32'b00000000000000000000000000100011
`define RVOPC_NOZ_SH 32'b00000000000000000001000000100011
`define RVOPC_NOZ_SW 32'b00000000000000000010000000100011
`define RVOPC_NOZ_FENCE 32'b00000000000000000000000000001111
`define RVOPC_NOZ_FENCE_I 32'b00000000000000000001000000001111
`define RVOPC_NOZ_ECALL 32'b00000000000000000000000001110011
`define RVOPC_NOZ_EBREAK 32'b00000000000100000000000001110011
`define RVOPC_NOZ_CSRRW 32'b00000000000000000001000001110011
`define RVOPC_NOZ_CSRRS 32'b00000000000000000010000001110011
`define RVOPC_NOZ_CSRRC 32'b00000000000000000011000001110011
`define RVOPC_NOZ_CSRRWI 32'b00000000000000000101000001110011
`define RVOPC_NOZ_CSRRSI 32'b00000000000000000110000001110011
`define RVOPC_NOZ_CSRRCI 32'b00000000000000000111000001110011
`define RVOPC_NOZ_SYSTEM 32'b00000000000000000000000001110011
// Non-RV32I instructions for Zcb:
`define RVOPC_NOZ_MUL 32'b00000010000000000000000000110011
`define RVOPC_NOZ_SEXT_B 32'b01100000010000000001000000010011
`define RVOPC_NOZ_SEXT_H 32'b01100000010100000001000000010011
`define RVOPC_NOZ_ZEXT_H 32'b00001000000000000100000000110011
// Non-RV32I instructions for Zclsd:
`define RVOPC_NOZ_LD 32'b00000000000000000011000000000011
`define RVOPC_NOZ_SD 32'b00000000000000000011000000100011
`endif