feat: publish FreeRTOS C FC05 card
This commit is contained in:
+269
@@ -0,0 +1,269 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_alu #(
|
||||
`include "hazard3_config.vh"
|
||||
,
|
||||
`include "hazard3_width_const.vh"
|
||||
) (
|
||||
input wire [W_ALUOP-1:0] aluop,
|
||||
input wire [6:0] funct7_32b,
|
||||
input wire [2:0] funct3_32b,
|
||||
input wire [W_DATA-1:0] op_a,
|
||||
input wire [W_DATA-1:0] op_b,
|
||||
output reg [W_DATA-1:0] result,
|
||||
output wire cmp
|
||||
);
|
||||
|
||||
`include "hazard3_ops.vh"
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Fiddle around with add/sub, comparisons etc (all related).
|
||||
|
||||
wire sub = !(aluop == ALUOP_ADD || (|EXTENSION_ZBA && aluop == ALUOP_SHXADD));
|
||||
|
||||
wire inv_op_b = sub && !(
|
||||
aluop == ALUOP_AND || aluop == ALUOP_OR || aluop == ALUOP_XOR || aluop == ALUOP_RS2
|
||||
);
|
||||
|
||||
wire [W_DATA-1:0] op_a_shifted =
|
||||
|EXTENSION_ZBA && aluop == ALUOP_SHXADD ? (
|
||||
!funct3_32b[2] ? op_a << 1 :
|
||||
!funct3_32b[1] ? op_a << 2 : op_a << 3
|
||||
) : op_a;
|
||||
|
||||
wire [W_DATA-1:0] op_b_inv = op_b ^ {W_DATA{inv_op_b}};
|
||||
|
||||
wire [W_DATA-1:0] sum = op_a_shifted + op_b_inv + {{W_DATA-1{1'b0}}, sub};
|
||||
wire [W_DATA-1:0] op_xor = op_a ^ op_b;
|
||||
|
||||
wire cmp_is_unsigned = aluop == ALUOP_LTU ||
|
||||
|EXTENSION_ZBB && aluop == ALUOP_MAXU ||
|
||||
|EXTENSION_ZBB && aluop == ALUOP_MINU;
|
||||
|
||||
wire lt = op_a[W_DATA-1] == op_b[W_DATA-1] ? sum[W_DATA-1] :
|
||||
cmp_is_unsigned ? op_b[W_DATA-1] : op_a[W_DATA-1] ;
|
||||
|
||||
assign cmp = aluop == ALUOP_SUB ? |op_xor : lt;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Separate units for shift, ctz etc
|
||||
|
||||
wire [W_DATA-1:0] shift_dout;
|
||||
wire shift_right_nleft =
|
||||
aluop == ALUOP_SRL ||
|
||||
aluop == ALUOP_SRA ||
|
||||
(|EXTENSION_ZBB && aluop == ALUOP_ROR ) ||
|
||||
(|EXTENSION_ZBS && aluop == ALUOP_BEXT ) ||
|
||||
(|EXTENSION_XH3BEXTM && aluop == ALUOP_BEXTM);
|
||||
|
||||
wire shift_arith = aluop == ALUOP_SRA;
|
||||
wire shift_rotate = |EXTENSION_ZBB & (aluop == ALUOP_ROR || aluop == ALUOP_ROL);
|
||||
|
||||
hazard3_shift_barrel #(
|
||||
`include "hazard3_config_inst.vh"
|
||||
) shifter (
|
||||
.din (op_a),
|
||||
.shamt (op_b[4:0]),
|
||||
.right_nleft (shift_right_nleft),
|
||||
.rotate (shift_rotate),
|
||||
.arith (shift_arith),
|
||||
.dout (shift_dout)
|
||||
);
|
||||
|
||||
reg [W_DATA-1:0] op_a_rev;
|
||||
always @ (*) begin: rev_op_a
|
||||
integer i;
|
||||
for (i = 0; i < W_DATA; i = i + 1) begin
|
||||
op_a_rev[i] = op_a[W_DATA - 1 - i];
|
||||
end
|
||||
end
|
||||
|
||||
// "leading" means starting at MSB. This is an LSB-first priority encoder, so
|
||||
// "leading" is reversed and "trailing" is not.
|
||||
wire [W_DATA-1:0] ctz_search_mask = aluop == ALUOP_CLZ ? op_a_rev : op_a;
|
||||
wire [W_SHAMT:0] ctz_clz;
|
||||
|
||||
hazard3_priority_encode #(
|
||||
.W_REQ (W_DATA),
|
||||
.HIGHEST_WINS (0)
|
||||
) ctz_priority_encode (
|
||||
.req (ctz_search_mask),
|
||||
.gnt (ctz_clz[W_SHAMT-1:0])
|
||||
);
|
||||
// Special case: all-zeroes returns XLEN
|
||||
assign ctz_clz[W_SHAMT] = ~|op_a;
|
||||
|
||||
reg [W_SHAMT:0] cpop;
|
||||
always @ (*) begin: cpop_count
|
||||
integer i;
|
||||
cpop = {W_SHAMT+1{1'b0}};
|
||||
for (i = 0; i < W_DATA; i = i + 1) begin
|
||||
cpop = cpop + {{W_SHAMT{1'b0}}, op_a[i]};
|
||||
end
|
||||
end
|
||||
|
||||
reg [2*W_DATA-1:0] clmul64;
|
||||
|
||||
always @ (*) begin: clmul_mul
|
||||
integer i;
|
||||
clmul64 = {2*W_DATA{1'b0}};
|
||||
for (i = 0; i < W_DATA; i = i + 1) begin
|
||||
clmul64 = clmul64 ^ (({{W_DATA{1'b0}}, op_a} << i) & {2*W_DATA{op_b[i]}});
|
||||
end
|
||||
end
|
||||
|
||||
// funct3: 1=clmul, 2=clmulr, 3=clmulh, never 0.
|
||||
wire [W_DATA-1:0] clmul =
|
||||
!funct3_32b[1] ? clmul64[31: 0] :
|
||||
!funct3_32b[0] ? clmul64[62:31] : clmul64[63:32];
|
||||
|
||||
reg [W_DATA-1:0] zip;
|
||||
reg [W_DATA-1:0] unzip;
|
||||
always @ (*) begin: do_zip_unzip
|
||||
integer i;
|
||||
for (i = 0; i < W_DATA; i = i + 1) begin
|
||||
zip[i] = op_a[{i[0], i[4:1]}]; // Alternate high/low halves
|
||||
unzip[i] = op_a[{i[3:0], i[4]}]; // All even then all odd
|
||||
end
|
||||
end
|
||||
|
||||
reg [W_DATA-1:0] xperm8;
|
||||
always @ (*) begin: do_xperm8
|
||||
integer i;
|
||||
for (i = 0; i < W_DATA; i = i + 8) begin
|
||||
if (|op_b[i + 2 +: 6]) begin
|
||||
xperm8[i +: 8] = 8'h00;
|
||||
end else begin
|
||||
xperm8[i +: 8] = op_a[8 * op_b[i +: 2] +: 8];
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
reg [W_DATA-1:0] xperm4;
|
||||
always @ (*) begin: do_xperm4
|
||||
integer i;
|
||||
for (i = 0; i < W_DATA; i = i + 4) begin
|
||||
if (op_b[i + 3]) begin
|
||||
xperm4[i +: 4] = 4'h0;
|
||||
end else begin
|
||||
xperm4[i +: 4] = op_a[4 * op_b[i +: 3] +: 4];
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Output mux, with simple operations inline
|
||||
|
||||
// iCE40: We can implement all bitwise ops with 1 LUT4/bit total, since each
|
||||
// result bit uses only two operand bits. Much better than feeding each into
|
||||
// main mux tree. Doesn't matter for big-LUT FPGAs or for implementations with
|
||||
// bitmanip extensions enabled.
|
||||
|
||||
reg [W_DATA-1:0] bitwise;
|
||||
|
||||
always @ (*) begin: bitwise_ops
|
||||
case (aluop[1:0])
|
||||
ALUOP_AND[1:0]: bitwise = op_a & op_b_inv;
|
||||
ALUOP_OR [1:0]: bitwise = op_a | op_b_inv;
|
||||
ALUOP_XOR[1:0]: bitwise = op_a ^ op_b_inv;
|
||||
ALUOP_RS2[1:0]: bitwise = op_b_inv;
|
||||
endcase
|
||||
end
|
||||
|
||||
wire [W_DATA-1:0] zbs_mask = {{W_DATA-1{1'b0}}, 1'b1} << op_b[W_SHAMT-1:0];
|
||||
|
||||
always @ (*) begin
|
||||
casez ({|EXTENSION_A, |EXTENSION_ZBA, |EXTENSION_ZBB, |EXTENSION_ZBC,
|
||||
|EXTENSION_ZBS, |EXTENSION_ZBKB, |EXTENSION_ZBKX, |EXTENSION_XH3BEXTM, aluop})
|
||||
// Base ISA
|
||||
{8'bzzzzzzzz, ALUOP_ADD }: result = sum;
|
||||
{8'bzzzzzzzz, ALUOP_SUB }: result = sum;
|
||||
{8'bzzzzzzzz, ALUOP_LT }: result = {{W_DATA-1{1'b0}}, lt};
|
||||
{8'bzzzzzzzz, ALUOP_LTU }: result = {{W_DATA-1{1'b0}}, lt};
|
||||
{8'bzzzzzzzz, ALUOP_SRL }: result = shift_dout;
|
||||
{8'bzzzzzzzz, ALUOP_SRA }: result = shift_dout;
|
||||
{8'bzzzzzzzz, ALUOP_SLL }: result = shift_dout;
|
||||
// A or Zbb (written this way to avoid case overlap)
|
||||
{8'b1zzzzzzz, ALUOP_MAX },
|
||||
{8'b0z1zzzzz, ALUOP_MAX }: result = lt ? op_b : op_a;
|
||||
{8'b1zzzzzzz, ALUOP_MIN },
|
||||
{8'b0z1zzzzz, ALUOP_MIN }: result = lt ? op_a : op_b;
|
||||
{8'b1zzzzzzz, ALUOP_MAXU },
|
||||
{8'b0z1zzzzz, ALUOP_MAXU }: result = lt ? op_b : op_a;
|
||||
{8'b1zzzzzzz, ALUOP_MINU },
|
||||
{8'b0z1zzzzz, ALUOP_MINU }: result = lt ? op_a : op_b;
|
||||
// Zba
|
||||
{8'bz1zzzzzz, ALUOP_SHXADD }: result = sum;
|
||||
// Zbb
|
||||
{8'bzz1zzzzz, ALUOP_ANDN }: result = bitwise;
|
||||
{8'bzz1zzzzz, ALUOP_ORN }: result = bitwise;
|
||||
{8'bzz1zzzzz, ALUOP_XNOR }: result = bitwise;
|
||||
{8'bzz1zzzzz, ALUOP_CLZ }: result = {{W_DATA-W_SHAMT-1{1'b0}}, ctz_clz};
|
||||
{8'bzz1zzzzz, ALUOP_CTZ }: result = {{W_DATA-W_SHAMT-1{1'b0}}, ctz_clz};
|
||||
{8'bzz1zzzzz, ALUOP_CPOP }: result = {{W_DATA-W_SHAMT-1{1'b0}}, cpop};
|
||||
{8'bzz1zzzzz, ALUOP_SEXT_B }: result = {{W_DATA-8{op_a[7]}}, op_a[7:0]};
|
||||
{8'bzz1zzzzz, ALUOP_SEXT_H }: result = {{W_DATA-16{op_a[15]}}, op_a[15:0]};
|
||||
{8'bzz1zzzzz, ALUOP_ZEXT_H }: result = {{W_DATA-16{1'b0}}, op_a[15:0]};
|
||||
{8'bzz1zzzzz, ALUOP_ORC_B }: result = {{8{|op_a[31:24]}}, {8{|op_a[23:16]}}, {8{|op_a[15:8]}}, {8{|op_a[7:0]}}};
|
||||
{8'bzz1zzzzz, ALUOP_REV8 }: result = {op_a[7:0], op_a[15:8], op_a[23:16], op_a[31:24]};
|
||||
{8'bzz1zzzzz, ALUOP_ROL }: result = shift_dout;
|
||||
{8'bzz1zzzzz, ALUOP_ROR }: result = shift_dout;
|
||||
// Zbc
|
||||
{8'bzzz1zzzz, ALUOP_CLMUL }: result = clmul;
|
||||
// Zbs
|
||||
{8'bzzzz1zzz, ALUOP_BCLR }: result = op_a & ~zbs_mask;
|
||||
{8'bzzzz1zzz, ALUOP_BSET }: result = op_a | zbs_mask;
|
||||
{8'bzzzz1zzz, ALUOP_BINV }: result = op_a ^ zbs_mask;
|
||||
{8'bzzzz1zzz, ALUOP_BEXT }: result = {{W_DATA-1{1'b0}}, shift_dout[0]};
|
||||
// Zbkb
|
||||
{8'bzzzzz1zz, ALUOP_PACK }: result = {op_b[15:0], op_a[15:0]};
|
||||
{8'bzzzzz1zz, ALUOP_PACKH }: result = {{W_DATA-16{1'b0}}, op_b[7:0], op_a[7:0]};
|
||||
{8'bzzzzz1zz, ALUOP_BREV8 }: result = {op_a_rev[7:0], op_a_rev[15:8], op_a_rev[23:16], op_a_rev[31:24]};
|
||||
{8'bzzzzz1zz, ALUOP_UNZIP }: result = unzip;
|
||||
{8'bzzzzz1zz, ALUOP_ZIP }: result = zip;
|
||||
// Zbkx
|
||||
{8'bzzzzzz1z, ALUOP_XPERM }: result = funct3_32b[2] ? xperm8 : xperm4;
|
||||
// Xh3bextm
|
||||
{8'bzzzzzzz1, ALUOP_BEXTM }: result = shift_dout & {24'h0, {~(8'hfe << funct7_32b[3:1])}};
|
||||
|
||||
default: result = bitwise;
|
||||
endcase
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Properties for base-ISA instructions
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
`ifndef RISCV_FORMAL
|
||||
// Really we're just interested in the shifts and comparisons, as these are
|
||||
// the nontrivial ones. However, easier to test everything!
|
||||
|
||||
wire clk;
|
||||
always @ (posedge clk) begin
|
||||
case(aluop)
|
||||
default: begin end
|
||||
ALUOP_ADD: assert(result == op_a + op_b);
|
||||
ALUOP_SUB: assert(result == op_a - op_b);
|
||||
ALUOP_LT: assert(result == $signed(op_a) < $signed(op_b));
|
||||
ALUOP_LTU: assert(result == op_a < op_b);
|
||||
ALUOP_AND: assert(result == (op_a & op_b));
|
||||
ALUOP_OR: assert(result == (op_a | op_b));
|
||||
ALUOP_XOR: assert(result == (op_a ^ op_b));
|
||||
ALUOP_SRL: assert(result == op_a >> op_b[4:0]);
|
||||
ALUOP_SRA: assert($signed(result) == $signed(op_a) >>> $signed(op_b[4:0]));
|
||||
ALUOP_SLL: assert(result == op_a << op_b[4:0]);
|
||||
endcase
|
||||
end
|
||||
`endif
|
||||
`endif
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+52
@@ -0,0 +1,52 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
`default_nettype none
|
||||
|
||||
// The branch decision path through the ALU is slow because:
|
||||
//
|
||||
// - Sees immediates and PC on its inputs, as well as regs
|
||||
// - Add/sub rather than just add (with complex decode of the sub condition)
|
||||
// - 2 extra mux layers in front of adder if Zba extension is enabled
|
||||
//
|
||||
// So there is sometimes timing benefit to a dedicated branch comparator.
|
||||
|
||||
module hazard3_branchcmp #(
|
||||
`include "hazard3_config.vh"
|
||||
,
|
||||
`include "hazard3_width_const.vh"
|
||||
) (
|
||||
input wire [31:0] cir,
|
||||
input wire [W_DATA-1:0] op_a,
|
||||
input wire [W_DATA-1:0] op_b,
|
||||
output wire cmp
|
||||
);
|
||||
|
||||
`include "hazard3_ops.vh"
|
||||
|
||||
wire [W_DATA-1:0] diff = op_a - op_b;
|
||||
|
||||
// funct3 instruction
|
||||
// ------------------
|
||||
// 000 BEQ
|
||||
// 001 BNE
|
||||
// 100 BLT
|
||||
// 101 BGE
|
||||
// 110 BLTU
|
||||
// 111 BGEU
|
||||
|
||||
wire cmp_is_unsigned = cir[13];
|
||||
|
||||
wire lt = op_a[W_DATA-1] == op_b[W_DATA-1] ? diff[W_DATA-1] :
|
||||
cmp_is_unsigned ? op_b[W_DATA-1] :
|
||||
op_a[W_DATA-1] ;
|
||||
|
||||
assign cmp = cir[14] ? lt : op_a != op_b;
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+162
@@ -0,0 +1,162 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// MUL-only (cfg: MUL_FAST) and MUL/MULH/MULHU/MULHSU (cfg: MUL_FAST &&
|
||||
// MULH_FAST) are handled by different circuits. In either case it's a simple
|
||||
// behavioural multiply, and we rely on inference to get good performance on
|
||||
// FPGA.
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_mul_fast #(
|
||||
`include "hazard3_config.vh"
|
||||
,
|
||||
`include "hazard3_width_const.vh"
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
|
||||
input wire [W_MULOP-1:0] op,
|
||||
input wire op_vld,
|
||||
input wire [W_DATA-1:0] op_a,
|
||||
input wire [W_DATA-1:0] op_b,
|
||||
|
||||
output wire [W_DATA-1:0] result,
|
||||
output reg result_vld
|
||||
);
|
||||
|
||||
`include "hazard3_ops.vh"
|
||||
|
||||
localparam XLEN = W_DATA;
|
||||
|
||||
//synthesis translate_off
|
||||
generate if (MULH_FAST && !MUL_FAST) begin: err_require_mul_fast_for_mulh
|
||||
initial $fatal("%m: MULH_FAST requires that MUL_FAST is also set.");
|
||||
end endgenerate
|
||||
generate if (MUL_FASTER && !MUL_FAST) begin: err_require_mul_fast_for_faster
|
||||
initial $fatal("%m: MUL_FASTER requires that MUL_FAST is also set.");
|
||||
end endgenerate
|
||||
//synthesis translate_on
|
||||
|
||||
// Latency of 1:
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
result_vld <= 1'b0;
|
||||
end else begin
|
||||
result_vld <= op_vld;
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Fast MUL only
|
||||
|
||||
generate
|
||||
if (!MULH_FAST) begin: mul_only
|
||||
|
||||
// This pipestage is folded into the front of the DSP tiles on UP5k. Note the
|
||||
// intention is to register the bypassed core regs at the end of X (since
|
||||
// bypass is quite slow), then perform multiply combinatorially in stage M,
|
||||
// and mux into MW result register.
|
||||
|
||||
reg [XLEN-1:0] op_a_r;
|
||||
reg [XLEN-1:0] op_b_r;
|
||||
|
||||
if (MUL_FASTER) begin: op_passthrough
|
||||
always @ (*) begin
|
||||
op_a_r = op_a;
|
||||
op_b_r = op_b;
|
||||
end
|
||||
end else begin: op_register
|
||||
always @ (posedge clk) begin
|
||||
if (op_vld) begin
|
||||
op_a_r <= op_a;
|
||||
op_b_r <= op_b;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// This should be inferred as 3 DSP tiles on UP5k:
|
||||
//
|
||||
// 1. Register then multiply a[15: 0] and b[15: 0]
|
||||
// 2. Register then multiply a[31:16] and b[15: 0], then directly add output of 1
|
||||
// 3. Register then multiply a[15: 0] and b[31:16], then directly add output of 2
|
||||
//
|
||||
// So there is quite a long path (1x 16-bit multiply, then 2x 16-bit add). On
|
||||
// other platforms you may just end up with a pile of gates.
|
||||
|
||||
`ifndef RISCV_FORMAL_ALTOPS
|
||||
|
||||
assign result = op_a_r * op_b_r;
|
||||
|
||||
`else
|
||||
|
||||
// riscv-formal can use a simpler function, since it's just confirming the
|
||||
// result is correctly hooked up.
|
||||
assign result = result_vld ? (op_a_r + op_b_r) ^ 32'h5876063e : 32'hdeadbeef;
|
||||
|
||||
`endif
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Fast MUL/MULH/MULHU/MULHSU
|
||||
|
||||
end else begin: mul_and_mulh
|
||||
|
||||
reg [XLEN-1:0] op_a_r;
|
||||
reg [XLEN-1:0] op_b_r;
|
||||
reg [W_MULOP-1:0] op_r;
|
||||
|
||||
if (MUL_FASTER) begin: op_passthrough
|
||||
always @ (*) begin
|
||||
op_a_r = op_a;
|
||||
op_b_r = op_b;
|
||||
op_r = op;
|
||||
end
|
||||
end else begin: op_register
|
||||
always @ (posedge clk) begin
|
||||
if (op_vld) begin
|
||||
op_a_r <= op_a;
|
||||
op_b_r <= op_b;
|
||||
op_r <= op;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
wire op_a_signed = op_r == M_OP_MULH || op_r == M_OP_MULHSU;
|
||||
wire op_b_signed = op_r == M_OP_MULH;
|
||||
|
||||
wire [2*XLEN-1:0] op_a_sext = {
|
||||
{XLEN{op_a_r[XLEN - 1] && op_a_signed}},
|
||||
op_a_r
|
||||
};
|
||||
|
||||
wire [2*XLEN-1:0] op_b_sext = {
|
||||
{XLEN{op_b_r[XLEN - 1] && op_b_signed}},
|
||||
op_b_r
|
||||
};
|
||||
|
||||
wire [2*XLEN-1:0] result_full = op_a_sext * op_b_sext;
|
||||
|
||||
`ifndef RISCV_FORMAL_ALTOPS
|
||||
|
||||
assign result = op_r == M_OP_MUL ? result_full[0 +: XLEN] : result_full[XLEN +: XLEN];
|
||||
|
||||
`else
|
||||
|
||||
assign result =
|
||||
op_r == M_OP_MULH ? (op_a_r + op_b_r) ^ 32'hf6583fb7 :
|
||||
op_r == M_OP_MULHSU ? (op_a_r - op_b_r) ^ 32'hecfbe137 :
|
||||
op_r == M_OP_MULHU ? (op_a_r + op_b_r) ^ 32'h949ce5e8 :
|
||||
op_r == M_OP_MUL ? (op_a_r + op_b_r) ^ 32'h5876063e : 32'hdeadbeef;
|
||||
|
||||
`endif
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+294
@@ -0,0 +1,294 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Combined multiply/divide/modulo circuit. All operations performed at 1 bit
|
||||
// per clock; aiming for minimal resource usage on iCE40 FPGA. Optionally the
|
||||
// circuit can be unrolled for slightly higher performance.
|
||||
//
|
||||
// When op_kill is high, the current calculation halts immediately. op_vld can
|
||||
// be asserted on the same cycle, and the new calculation begins without
|
||||
// delay, regardless of op_rdy. This may be used by the processor on e.g.
|
||||
// mispredict or trap.
|
||||
//
|
||||
// The actual multiply/divide hardware is unsigned. We handle signedness at
|
||||
// input/output.
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_muldiv_seq #(
|
||||
`include "hazard3_config.vh"
|
||||
,
|
||||
`include "hazard3_width_const.vh"
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
input wire [W_MULOP-1:0] op,
|
||||
input wire op_vld,
|
||||
output wire op_rdy,
|
||||
input wire op_kill,
|
||||
input wire [W_DATA-1:0] op_a,
|
||||
input wire [W_DATA-1:0] op_b,
|
||||
|
||||
output wire [W_DATA-1:0] result_h, // mulh* or rem*
|
||||
output wire [W_DATA-1:0] result_l, // mul or div*
|
||||
output wire result_vld
|
||||
);
|
||||
|
||||
`include "hazard3_ops.vh"
|
||||
|
||||
//synthesis translate_off
|
||||
generate if (|(MULDIV_UNROLL & (MULDIV_UNROLL - 1)) || ~|MULDIV_UNROLL) begin: err_pow2
|
||||
initial $fatal("%m: MULDIV_UNROLL must be a positive power of 2");
|
||||
end endgenerate
|
||||
//synthesis translate_on
|
||||
|
||||
localparam XLEN = W_DATA;
|
||||
parameter W_CTR = $clog2(XLEN + 1);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Operation decode, operand sign adjustment
|
||||
|
||||
// On the first cycle, op_a and op_b go straight through to the accumulator
|
||||
// and the divisor/multiplicand register. They are then adjusted in-place
|
||||
// on the next cycle. This allows the same circuits to be reused for sign
|
||||
// adjustment before output (and helps input timing).
|
||||
|
||||
reg [W_MULOP-1:0] op_r;
|
||||
reg [2*XLEN-1:0] accum;
|
||||
reg [XLEN-1:0] op_b_r;
|
||||
reg op_a_neg_r;
|
||||
reg op_b_neg_r;
|
||||
|
||||
wire op_a_signed =
|
||||
op_r == M_OP_MULH ||
|
||||
op_r == M_OP_MULHSU ||
|
||||
op_r == M_OP_DIV ||
|
||||
op_r == M_OP_REM;
|
||||
|
||||
wire op_b_signed =
|
||||
op_r == M_OP_MULH ||
|
||||
op_r == M_OP_DIV ||
|
||||
op_r == M_OP_REM;
|
||||
|
||||
wire op_a_neg = op_a_signed && accum[XLEN-1];
|
||||
wire op_b_neg = op_b_signed && op_b_r[XLEN-1];
|
||||
|
||||
// Non-divide parts of the circuit should be constant-folded if all the MUL
|
||||
// operations are handled by the fast multiplier
|
||||
|
||||
wire is_div = op_r[2] || (MUL_FAST && MULH_FAST);
|
||||
|
||||
// Controls for modifying sign of all/part of accumulator
|
||||
wire accum_neg_l;
|
||||
wire accum_inv_h;
|
||||
wire accum_incr_h;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Arithmetic circuit
|
||||
|
||||
// Combinatorials:
|
||||
reg [2*XLEN-1:0] accum_next;
|
||||
reg [2*XLEN-1:0] addend;
|
||||
reg [2*XLEN-1:0] shift_tmp;
|
||||
reg [2*XLEN-1:0] addsub_tmp;
|
||||
reg neg_l_borrow;
|
||||
|
||||
always @ (*) begin: alu
|
||||
integer i;
|
||||
// Multiply/divide iteration layers
|
||||
accum_next = accum;
|
||||
addend = {2*XLEN{1'b0}};
|
||||
addsub_tmp = {2*XLEN{1'b0}};
|
||||
neg_l_borrow = 1'b0;
|
||||
for (i = 0; i < MULDIV_UNROLL; i = i + 1) begin
|
||||
addend = {is_div && |op_b_r, op_b_r, {XLEN-1{1'b0}}};
|
||||
shift_tmp = is_div ? accum_next : accum_next >> 1;
|
||||
addsub_tmp = shift_tmp + addend;
|
||||
accum_next = (is_div ? !addsub_tmp[2 * XLEN - 1] : accum_next[0]) ?
|
||||
addsub_tmp : shift_tmp;
|
||||
if (is_div)
|
||||
accum_next = {accum_next[2*XLEN-2:0], !addsub_tmp[2 * XLEN - 1]};
|
||||
end
|
||||
// Alternative path for negation of all/part of accumulator
|
||||
if (accum_neg_l)
|
||||
{neg_l_borrow, accum_next[XLEN-1:0]} = {~accum[XLEN-1:0]} + 1'b1;
|
||||
if (accum_incr_h || accum_inv_h)
|
||||
accum_next[XLEN +: XLEN] = (accum[XLEN +: XLEN] ^ {XLEN{accum_inv_h}})
|
||||
+ {{XLEN-1{1'b0}}, accum_incr_h};
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Main state machine
|
||||
|
||||
reg sign_preadj_done;
|
||||
reg [W_CTR-1:0] ctr;
|
||||
reg sign_postadj_done;
|
||||
reg sign_postadj_carry;
|
||||
|
||||
localparam CTR_TOP = XLEN[W_CTR-1:0];
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
ctr <= {W_CTR{1'b0}};
|
||||
sign_preadj_done <= 1'b1;
|
||||
sign_postadj_done <= 1'b1;
|
||||
sign_postadj_carry <= 1'b0;
|
||||
op_r <= {W_MULOP{1'b0}};
|
||||
op_a_neg_r <= 1'b0;
|
||||
op_b_neg_r <= 1'b0;
|
||||
op_b_r <= {XLEN{1'b0}};
|
||||
accum <= {XLEN*2{1'b0}};
|
||||
end else if (op_kill || (op_vld && op_rdy)) begin
|
||||
// Initialise circuit with operands + state
|
||||
ctr <= op_vld ? CTR_TOP : {W_CTR{1'b0}};
|
||||
sign_preadj_done <= !op_vld;
|
||||
sign_postadj_done <= !op_vld;
|
||||
sign_postadj_carry <= 1'b0;
|
||||
op_r <= op;
|
||||
op_b_r <= op_b;
|
||||
accum <= {{XLEN{1'b0}}, op_a};
|
||||
end else if (!sign_preadj_done) begin
|
||||
// Pre-adjust sign if necessary, else perform first iteration immediately
|
||||
op_a_neg_r <= op_a_neg;
|
||||
op_b_neg_r <= op_b_neg;
|
||||
sign_preadj_done <= 1'b1;
|
||||
if (accum_neg_l || (op_b_neg ^ is_div)) begin
|
||||
if (accum_neg_l)
|
||||
accum[0 +: XLEN] <= accum_next[0 +: XLEN];
|
||||
if (op_b_neg ^ is_div)
|
||||
op_b_r <= -op_b_r;
|
||||
end else begin
|
||||
ctr <= ctr - MULDIV_UNROLL[W_CTR-1:0];
|
||||
accum <= accum_next;
|
||||
end
|
||||
end else if (|ctr) begin
|
||||
ctr <= ctr - MULDIV_UNROLL[W_CTR-1:0];
|
||||
accum <= accum_next;
|
||||
end else if (!sign_postadj_done || sign_postadj_carry) begin
|
||||
sign_postadj_done <= 1'b1;
|
||||
if (accum_inv_h || accum_incr_h)
|
||||
accum[XLEN +: XLEN] <= accum_next[XLEN +: XLEN];
|
||||
if (accum_neg_l) begin
|
||||
accum[0 +: XLEN] <= accum_next[0 +: XLEN];
|
||||
if (!is_div) begin
|
||||
sign_postadj_carry <= neg_l_borrow;
|
||||
sign_postadj_done <= !neg_l_borrow;
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Sign adjustment control
|
||||
|
||||
// Pre-adjustment: for any a, b we want |a|, |b|. Note that the magnitude of any
|
||||
// 32-bit signed integer is representable by a 32-bit unsigned integer.
|
||||
|
||||
// Post-adjustment for division:
|
||||
// We seek q, r to satisfy a = b * q + r, where a and b are given,
|
||||
// and |r| < |b|. One way to do this is if
|
||||
// sgn(r) = sgn(a)
|
||||
// sgn(q) = sgn(a) ^ sgn(b)
|
||||
// This has additional nice properties like
|
||||
// -(a / b) = (-a) / b = a / (-b)
|
||||
|
||||
// Post-adjustment for multiplication:
|
||||
// We have calculated the 2*XLEN result of |a| * |b|.
|
||||
// Negate the entire accumulator if sgn(a) ^ sgn(b).
|
||||
// This is done in two steps (to share div/mod circuit, and avoid 64-bit carry):
|
||||
// - Negate lower half of accumulator, and invert upper half
|
||||
// - Increment upper half if lower half carried
|
||||
|
||||
wire do_postadj = ~|{ctr, sign_postadj_done};
|
||||
wire op_signs_differ = op_a_neg_r ^ op_b_neg_r;
|
||||
|
||||
assign accum_neg_l =
|
||||
!sign_preadj_done && op_a_neg ||
|
||||
do_postadj && !sign_postadj_carry && op_signs_differ && !(is_div && ~|op_b_r);
|
||||
|
||||
assign {accum_incr_h, accum_inv_h} =
|
||||
do_postadj && is_div && op_a_neg_r ? 2'b11 :
|
||||
do_postadj && !is_div && op_signs_differ && !sign_postadj_carry ? 2'b01 :
|
||||
do_postadj && !is_div && op_signs_differ && sign_postadj_carry ? 2'b10 :
|
||||
2'b00 ;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Outputs
|
||||
|
||||
assign op_rdy = ~|{ctr, accum_neg_l, accum_incr_h, accum_inv_h};
|
||||
assign result_vld = op_rdy;
|
||||
|
||||
`ifndef RISCV_FORMAL_ALTOPS
|
||||
|
||||
assign {result_h, result_l} = accum;
|
||||
|
||||
`else
|
||||
|
||||
// Provide arithmetically simpler alternative operations, to speed up formal checks
|
||||
always assert(XLEN == 32);
|
||||
|
||||
reg [XLEN-1:0] fml_a_saved;
|
||||
reg [XLEN-1:0] fml_b_saved;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
fml_a_saved <= {XLEN{1'b0}};
|
||||
fml_b_saved <= {XLEN{1'b0}};
|
||||
end else if (op_vld && op_rdy) begin
|
||||
fml_a_saved <= op_a;
|
||||
fml_b_saved <= op_b;
|
||||
end
|
||||
end
|
||||
|
||||
assign result_h =
|
||||
op_r == M_OP_MULH ? (fml_a_saved + fml_b_saved) ^ 32'hf6583fb7 :
|
||||
op_r == M_OP_MULHSU ? (fml_a_saved - fml_b_saved) ^ 32'hecfbe137 :
|
||||
op_r == M_OP_MULHU ? (fml_a_saved + fml_b_saved) ^ 32'h949ce5e8 :
|
||||
op_r == M_OP_REM ? (fml_a_saved - fml_b_saved) ^ 32'h8da68fa5 :
|
||||
op_r == M_OP_REMU ? (fml_a_saved - fml_b_saved) ^ 32'h3138d0e1 : 32'hdeadbeef;
|
||||
|
||||
assign result_l =
|
||||
op_r == M_OP_MUL ? (fml_a_saved + fml_b_saved) ^ 32'h5876063e :
|
||||
op_r == M_OP_DIV ? (fml_a_saved - fml_b_saved) ^ 32'h7f8529ec :
|
||||
op_r == M_OP_DIVU ? (fml_a_saved - fml_b_saved) ^ 32'h10e8fd70 : 32'hdeadbeef;
|
||||
|
||||
`endif
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Interface properties
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
|
||||
always @ (posedge clk) if (rst_n && $past(rst_n)) begin: properties
|
||||
integer i;
|
||||
reg alive;
|
||||
|
||||
if ($past(op_rdy && !op_vld))
|
||||
assert(op_rdy);
|
||||
|
||||
if (result_vld && $past(result_vld) && !$past(op_kill))
|
||||
assert($stable({result_h, result_l}));
|
||||
|
||||
// Kill will halt an in-progress operation, but a new operation may be
|
||||
// asserted simultaneously with kill.
|
||||
if ($past(op_kill))
|
||||
assert(op_rdy == !$past(op_vld));
|
||||
|
||||
// We should be periodically ready (liveness property), unless new operations
|
||||
// are forced in immediately, simultaneous with a kill, in which case there
|
||||
// is no intermediate ready state.
|
||||
alive = op_rdy || (op_kill && op_vld);
|
||||
for (i = 1; i <= XLEN / MULDIV_UNROLL + 3; i = i + 1)
|
||||
alive = alive || $past(op_rdy || (op_kill && op_vld), i);
|
||||
assert(alive);
|
||||
end
|
||||
|
||||
`endif
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,31 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// req: one-hot bitmap
|
||||
// idx: index of the sole set bit in req
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_onehot_encode #(
|
||||
parameter W_REQ = 16,
|
||||
parameter W_GNT = $clog2(W_REQ) // do not modify
|
||||
) (
|
||||
input wire [W_REQ-1:0] req,
|
||||
output reg [W_GNT-1:0] gnt
|
||||
);
|
||||
|
||||
always @ (*) begin: encode
|
||||
reg [W_GNT:0] i;
|
||||
gnt = {W_GNT{1'b0}};
|
||||
for (i = 0; i < W_REQ; i = i + 1) begin
|
||||
gnt = gnt | ({W_GNT{req[i[W_GNT-1:0]]}} & i[W_GNT-1:0]);
|
||||
end
|
||||
end
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,33 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// req: bitmap
|
||||
// idx: bitmap with all bits clear except the least- (HIGHEST_WINS=0) or
|
||||
// most- (HIGHEST_WINS=1) significant set bit in req.
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_onehot_priority #(
|
||||
parameter W_REQ = 16,
|
||||
parameter HIGHEST_WINS = 0
|
||||
) (
|
||||
input wire [W_REQ-1:0] req,
|
||||
output reg [W_REQ-1:0] gnt
|
||||
);
|
||||
|
||||
always @ (*) begin: select
|
||||
integer i;
|
||||
for (i = 0; i < W_REQ; i = i + 1) begin
|
||||
gnt[i] = req[i] && ~|(req & (
|
||||
HIGHEST_WINS ? ~({W_REQ{1'b1}} >> (W_REQ - 1 - i)) : ~({W_REQ{1'b1}} << i)
|
||||
));
|
||||
end
|
||||
end
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,77 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// req: bitmap of requests
|
||||
// priority: packed array of dynamic priority level of each request
|
||||
// gnt: one-hot bitmap with the highest-priority request.
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_onehot_priority_dynamic #(
|
||||
parameter W_REQ = 8,
|
||||
parameter N_PRIORITIES = 2,
|
||||
parameter PRIORITY_HIGHEST_WINS = 1, // If 1, numerically highest level has greatest priority.
|
||||
// Otherwise, numerically lowest wins.
|
||||
parameter TIEBREAK_HIGHEST_WINS = 0, // If 1, highest-numbered request at the highest priority
|
||||
// level wins the tiebreak. Otherwise, lowest-numbered.
|
||||
// Do not modify:
|
||||
parameter W_PRIORITY = $clog2(N_PRIORITIES)
|
||||
) (
|
||||
input wire [W_REQ*W_PRIORITY-1:0] pri,
|
||||
input wire [W_REQ-1:0] req,
|
||||
output wire [W_REQ-1:0] gnt
|
||||
);
|
||||
|
||||
// 1. Stratify requests according to their level
|
||||
reg [W_REQ-1:0] req_stratified [0:N_PRIORITIES-1];
|
||||
reg [N_PRIORITIES-1:0] level_has_req;
|
||||
|
||||
always @ (*) begin: stratify
|
||||
reg signed [31:0] i, j;
|
||||
for (i = 0; i < N_PRIORITIES; i = i + 1) begin
|
||||
for (j = 0; j < W_REQ; j = j + 1) begin
|
||||
req_stratified[i][j] = req[j] &&
|
||||
pri[W_PRIORITY * j +: W_PRIORITY] == i[W_PRIORITY-1:0];
|
||||
end
|
||||
level_has_req[i] = |req_stratified[i];
|
||||
end
|
||||
end
|
||||
|
||||
// 2. Select the highest level with active requests
|
||||
wire [N_PRIORITIES-1:0] active_layer_sel;
|
||||
|
||||
hazard3_onehot_priority #(
|
||||
.W_REQ (N_PRIORITIES),
|
||||
.HIGHEST_WINS (PRIORITY_HIGHEST_WINS)
|
||||
) prisel_layer (
|
||||
.req (level_has_req),
|
||||
.gnt (active_layer_sel)
|
||||
);
|
||||
|
||||
// 3. Mask only those requests at this level
|
||||
reg [W_REQ-1:0] reqs_from_highest_layer;
|
||||
|
||||
always @ (*) begin: mux_reqs_by_layer
|
||||
integer i;
|
||||
reqs_from_highest_layer = {W_REQ{1'b0}};
|
||||
for (i = 0; i < N_PRIORITIES; i = i + 1)
|
||||
reqs_from_highest_layer = reqs_from_highest_layer |
|
||||
(req_stratified[i] & {W_REQ{active_layer_sel[i]}});
|
||||
end
|
||||
|
||||
// 4. Do a standard priority select on those requests as a tie break
|
||||
hazard3_onehot_priority #(
|
||||
.W_REQ (W_REQ),
|
||||
.HIGHEST_WINS (TIEBREAK_HIGHEST_WINS)
|
||||
) prisel_tiebreak (
|
||||
.req (reqs_from_highest_layer),
|
||||
.gnt (gnt)
|
||||
);
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,41 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// req: bitmap
|
||||
// gnt: index of least set bit (HIGHEST_WINS=0) or most set bit (HIGHEST_WINS=1)
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_priority_encode #(
|
||||
parameter W_REQ = 16,
|
||||
parameter HIGHEST_WINS = 0,
|
||||
parameter W_GNT = $clog2(W_REQ) // do not modify
|
||||
) (
|
||||
input wire [W_REQ-1:0] req,
|
||||
output wire [W_GNT-1:0] gnt
|
||||
);
|
||||
|
||||
wire [W_REQ-1:0] gnt_onehot;
|
||||
|
||||
hazard3_onehot_priority #(
|
||||
.W_REQ (W_REQ),
|
||||
.HIGHEST_WINS (HIGHEST_WINS)
|
||||
) priority_u (
|
||||
.req (req),
|
||||
.gnt (gnt_onehot)
|
||||
);
|
||||
|
||||
hazard3_onehot_encode #(
|
||||
.W_REQ (W_REQ)
|
||||
) encode_u (
|
||||
.req (gnt_onehot),
|
||||
.gnt (gnt)
|
||||
);
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+66
@@ -0,0 +1,66 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Implement the three shifts (left logical, right logical, right arithmetic)
|
||||
// using a single log-type barrel shifter. Around 240 LUTs for 32 bits.
|
||||
// (7 layers of 32 2-input muxes, some extra LUTs and LUT inputs used for arith)
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_shift_barrel #(
|
||||
`include "hazard3_config.vh"
|
||||
,
|
||||
`include "hazard3_width_const.vh"
|
||||
) (
|
||||
input wire [W_DATA-1:0] din,
|
||||
input wire [W_SHAMT-1:0] shamt,
|
||||
input wire right_nleft,
|
||||
input wire rotate,
|
||||
input wire arith,
|
||||
output reg [W_DATA-1:0] dout
|
||||
);
|
||||
|
||||
reg [W_DATA-1:0] din_rev;
|
||||
reg [W_DATA-1:0] shift_accum;
|
||||
reg sext; // haha
|
||||
|
||||
always @ (*) begin: shift
|
||||
integer i;
|
||||
|
||||
for (i = 0; i < W_DATA; i = i + 1)
|
||||
din_rev[i] = right_nleft ? din[W_DATA - 1 - i] : din[i];
|
||||
|
||||
sext = arith && din_rev[0];
|
||||
|
||||
shift_accum = din_rev;
|
||||
for (i = 0; i < W_SHAMT; i = i + 1) begin
|
||||
if (shamt[i]) begin
|
||||
shift_accum = (shift_accum << (1 << i)) |
|
||||
({W_DATA{sext}} & ~({W_DATA{1'b1}} << (1 << i))) |
|
||||
({W_DATA{rotate && |EXTENSION_ZBB}} & (shift_accum >> (W_DATA - (1 << i))));
|
||||
end
|
||||
end
|
||||
|
||||
for (i = 0; i < W_DATA; i = i + 1)
|
||||
dout[i] = right_nleft ? shift_accum[W_DATA - 1 - i] : shift_accum[i];
|
||||
end
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
always @ (*) begin
|
||||
if (right_nleft && arith && !rotate) begin: asr
|
||||
assert($signed(dout) == $signed(din) >>> $signed(shamt));
|
||||
end else if (right_nleft && !arith && !rotate) begin
|
||||
assert(dout == din >> shamt);
|
||||
end else if (!right_nleft && !arith && !rotate) begin
|
||||
assert(dout == din << shamt);
|
||||
end
|
||||
end
|
||||
`endif
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+65
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
# Quick reference model for sequential unsigned multiply/divide/modulo
|
||||
|
||||
def div_step(w, accum, divisor):
|
||||
sub_tmp = accum - (divisor << (w - 1))
|
||||
underflow = sub_tmp < 0
|
||||
if not underflow:
|
||||
accum = sub_tmp
|
||||
accum = (accum << 1) | (not underflow)
|
||||
return accum
|
||||
|
||||
def divmod(w, dividend, divisor, debug=True):
|
||||
accum = dividend
|
||||
for i in range(w):
|
||||
accum_prev = accum
|
||||
accum = div_step(w, accum, divisor)
|
||||
if debug:
|
||||
print("Step {:02d}: accum {:0{}x} -> {:0{}x}".format(
|
||||
i, accum_prev, int(w / 2), accum, int(w / 2)))
|
||||
return (accum >> w, accum & ((1 << w) - 1))
|
||||
|
||||
def mul_step(w, accum, multiplicand):
|
||||
add_en = accum & 1
|
||||
accum = accum >> 1
|
||||
if add_en:
|
||||
accum += (multiplicand << (w - 1))
|
||||
return accum
|
||||
|
||||
def mul(w, multiplicand, multiplier, debug=True):
|
||||
accum = multiplier
|
||||
for i in range(w):
|
||||
accum_prev = accum
|
||||
accum = mul_step(w, accum, multiplicand)
|
||||
if debug:
|
||||
print("Step {:02d}: accum {:0{}x} -> {:0{}x}".format(
|
||||
i, accum_prev, int(w / 2), accum, int(w / 2)))
|
||||
return (accum >> w, accum & ((1 << w) - 1))
|
||||
|
||||
def divtest(w=4):
|
||||
for i in range(2 ** w):
|
||||
for j in range(1, 2 ** w):
|
||||
gatemod, gatediv = divmod(w, i, j, debug=False)
|
||||
goldmod, golddiv = (i % j, i // j)
|
||||
print("{:02d} % {:02d} = {:02d} (gold {:02d}); ./. = {:02d} (gold {:02d})"
|
||||
.format(i, j, gatemod, goldmod, gatediv, golddiv))
|
||||
assert(gatemod == goldmod)
|
||||
assert(gatediv == golddiv)
|
||||
|
||||
def multest(w=4):
|
||||
for i in range(2 ** w):
|
||||
for j in range(2 ** w):
|
||||
gateh, gatel = mul(w, i, j, debug=False)
|
||||
gold = i * j
|
||||
goldl, goldh = (gold & ((1 << w) - 1), gold >> w)
|
||||
print("{:02d} * {:02d} = ({:02d} (gold {:02d}), {:02d} (gold {:02d})"
|
||||
.format(i, j, gateh, goldh, gatel, goldl))
|
||||
assert(gatel == goldl)
|
||||
assert(gateh == goldh)
|
||||
|
||||
if __name__ == "__main__":
|
||||
print("Test division:")
|
||||
divtest()
|
||||
print("Test multiplication:")
|
||||
multest()
|
||||
@@ -0,0 +1,200 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// APB-to-APB asynchronous bridge for connecting DTM to DM, in case DTM is in
|
||||
// a different clock domain (e.g. running directly from crystal to get a
|
||||
// fixed baud reference)
|
||||
//
|
||||
// Note this module depends on the hazard3_sync_1bit module (a flop-chain
|
||||
// synchroniser) which should be reimplemented for your FPGA/process.
|
||||
|
||||
`ifndef HAZARD3_REG_KEEP_ATTRIBUTE
|
||||
`define HAZARD3_REG_KEEP_ATTRIBUTE (* keep = 1'b1 *)
|
||||
`endif
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_apb_async_bridge #(
|
||||
parameter W_ADDR = 8,
|
||||
parameter W_DATA = 32,
|
||||
parameter N_SYNC_STAGES = 2
|
||||
) (
|
||||
// Resets assumed to be synchronised externally
|
||||
input wire clk_src,
|
||||
input wire rst_n_src,
|
||||
|
||||
input wire clk_dst,
|
||||
input wire rst_n_dst,
|
||||
|
||||
// APB port from Transport Module
|
||||
input wire src_psel,
|
||||
input wire src_penable,
|
||||
input wire src_pwrite,
|
||||
input wire [W_ADDR-1:0] src_paddr,
|
||||
input wire [W_DATA-1:0] src_pwdata,
|
||||
output wire [W_DATA-1:0] src_prdata,
|
||||
output wire src_pready,
|
||||
output wire src_pslverr,
|
||||
|
||||
// APB port to Debug Module
|
||||
output wire dst_psel,
|
||||
output wire dst_penable,
|
||||
output wire dst_pwrite,
|
||||
output wire [W_ADDR-1:0] dst_paddr,
|
||||
output wire [W_DATA-1:0] dst_pwdata,
|
||||
input wire [W_DATA-1:0] dst_prdata,
|
||||
input wire dst_pready,
|
||||
input wire dst_pslverr
|
||||
);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Clock-crossing registers
|
||||
|
||||
// We're using a modified req/ack handshake:
|
||||
//
|
||||
// - Initially both req and ack are low
|
||||
// - src asserts req high
|
||||
// - dst responds with ack high and begins transfer
|
||||
// - src deasserts req once it sees ack high
|
||||
// - dst deasserts ack once:
|
||||
// - transfer is complete *and*
|
||||
// - dst sees req deasserted
|
||||
// - Once src sees ack low, a new transfer can begin.
|
||||
//
|
||||
// A NRZI toggle handshake might be more appropriate, but can cause spurious
|
||||
// bus accesses when only one side of the link is reset.
|
||||
|
||||
`HAZARD3_REG_KEEP_ATTRIBUTE reg src_req;
|
||||
wire dst_req;
|
||||
|
||||
`HAZARD3_REG_KEEP_ATTRIBUTE reg dst_ack;
|
||||
wire src_ack;
|
||||
|
||||
// Note the launch registers are not resettable. We maintain setup/hold on
|
||||
// launch-to-capture paths thanks to the req/ack handshake. A stray reset
|
||||
// could violate this.
|
||||
//
|
||||
// The req/ack logic itself can be reset safely because the receiving domain
|
||||
// is protected from metastability by a 2FF synchroniser.
|
||||
|
||||
`HAZARD3_REG_KEEP_ATTRIBUTE reg [W_ADDR + W_DATA + 1 -1:0] src_paddr_pwdata_pwrite; // launch
|
||||
`HAZARD3_REG_KEEP_ATTRIBUTE reg [W_ADDR + W_DATA + 1 -1:0] dst_paddr_pwdata_pwrite; // capture
|
||||
|
||||
`HAZARD3_REG_KEEP_ATTRIBUTE reg [W_DATA + 1 -1:0] dst_prdata_pslverr; // launch
|
||||
`HAZARD3_REG_KEEP_ATTRIBUTE reg [W_DATA + 1 -1:0] src_prdata_pslverr; // capture
|
||||
|
||||
hazard3_sync_1bit #(
|
||||
.N_STAGES (N_SYNC_STAGES)
|
||||
) sync_req (
|
||||
.clk (clk_dst),
|
||||
.rst_n (rst_n_dst),
|
||||
.i (src_req),
|
||||
.o (dst_req)
|
||||
);
|
||||
|
||||
hazard3_sync_1bit #(
|
||||
.N_STAGES (N_SYNC_STAGES)
|
||||
) sync_ack (
|
||||
.clk (clk_src),
|
||||
.rst_n (rst_n_src),
|
||||
.i (dst_ack),
|
||||
.o (src_ack)
|
||||
);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// src state machine
|
||||
|
||||
reg src_waiting_for_downstream;
|
||||
reg src_pready_r;
|
||||
|
||||
always @ (posedge clk_src or negedge rst_n_src) begin
|
||||
if (!rst_n_src) begin
|
||||
src_req <= 1'b0;
|
||||
src_waiting_for_downstream <= 1'b0;
|
||||
src_prdata_pslverr <= {W_DATA + 1{1'b0}};
|
||||
src_pready_r <= 1'b1;
|
||||
end else if (src_waiting_for_downstream) begin
|
||||
if (src_req && src_ack) begin
|
||||
// Request was acknowledged, so deassert.
|
||||
src_req <= 1'b0;
|
||||
end else if (!(src_req || src_ack)) begin
|
||||
// Downstream transfer has finished, data is valid.
|
||||
src_pready_r <= 1'b1;
|
||||
src_waiting_for_downstream <= 1'b0;
|
||||
// Note this assignment is cross-domain (but data has been stable
|
||||
// for duration of ack synchronisation delay):
|
||||
src_prdata_pslverr <= dst_prdata_pslverr;
|
||||
end
|
||||
end else begin
|
||||
// paddr, pwdata and pwrite are all valid during the setup phase, and
|
||||
// APB defines the setup phase to always last one cycle and proceed
|
||||
// to access phase. So, we can ignore penable, and pready is ignored.
|
||||
if (src_psel) begin
|
||||
src_pready_r <= 1'b0;
|
||||
src_req <= 1'b1;
|
||||
src_waiting_for_downstream <= 1'b1;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// Bus request launch register is not resettable
|
||||
always @ (posedge clk_src) begin
|
||||
if (src_psel && !src_waiting_for_downstream)
|
||||
src_paddr_pwdata_pwrite <= {src_paddr, src_pwdata, src_pwrite};
|
||||
end
|
||||
|
||||
assign {src_prdata, src_pslverr} = src_prdata_pslverr;
|
||||
assign src_pready = src_pready_r;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// dst state machine
|
||||
|
||||
wire dst_bus_finish = dst_penable && dst_pready;
|
||||
reg dst_psel_r;
|
||||
reg dst_penable_r;
|
||||
|
||||
always @ (posedge clk_dst or negedge rst_n_dst) begin
|
||||
if (!rst_n_dst) begin
|
||||
dst_ack <= 1'b0;
|
||||
end else if (dst_req) begin
|
||||
dst_ack <= 1'b1;
|
||||
end else if (!dst_req && dst_ack && !dst_psel_r) begin
|
||||
dst_ack <= 1'b0;
|
||||
end
|
||||
end
|
||||
|
||||
always @ (posedge clk_dst or negedge rst_n_dst) begin
|
||||
if (!rst_n_dst) begin
|
||||
dst_psel_r <= 1'b0;
|
||||
dst_penable_r <= 1'b0;
|
||||
dst_paddr_pwdata_pwrite <= {W_ADDR + W_DATA + 1{1'b0}};
|
||||
end else if (dst_req && !dst_ack) begin
|
||||
dst_psel_r <= 1'b1;
|
||||
// Note this assignment is cross-domain. The src register has been
|
||||
// stable for the duration of the req sync delay.
|
||||
dst_paddr_pwdata_pwrite <= src_paddr_pwdata_pwrite;
|
||||
end else if (dst_psel_r && !dst_penable_r) begin
|
||||
dst_penable_r <= 1'b1;
|
||||
end else if (dst_bus_finish) begin
|
||||
dst_psel_r <= 1'b0;
|
||||
dst_penable_r <= 1'b0;
|
||||
end
|
||||
end
|
||||
|
||||
// Bus response launch register is not resettable
|
||||
always @ (posedge clk_dst) begin
|
||||
if (dst_bus_finish)
|
||||
dst_prdata_pslverr <= {dst_prdata, dst_pslverr};
|
||||
end
|
||||
|
||||
assign dst_psel = dst_psel_r;
|
||||
assign dst_penable = dst_penable_r;
|
||||
assign {dst_paddr, dst_pwdata, dst_pwrite} = dst_paddr_pwdata_pwrite;
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,41 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// The output is asserted asynchronously when the input is asserted,
|
||||
// but deasserted synchronously when clocked with the input deasserted.
|
||||
// Input and output are both active-low.
|
||||
//
|
||||
// This is a baseline implementation -- you should replace it with cells
|
||||
// specific to your FPGA/process
|
||||
|
||||
`ifndef HAZARD3_REG_KEEP_ATTRIBUTE
|
||||
`define HAZARD3_REG_KEEP_ATTRIBUTE (* keep = 1'b1 *)
|
||||
`endif
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_reset_sync #(
|
||||
parameter N_STAGES = 2 // Should be >= 2
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n_in,
|
||||
output wire rst_n_out
|
||||
);
|
||||
|
||||
`HAZARD3_REG_KEEP_ATTRIBUTE reg [N_STAGES-1:0] delay;
|
||||
|
||||
always @ (posedge clk or negedge rst_n_in)
|
||||
if (!rst_n_in)
|
||||
delay <= {N_STAGES{1'b0}};
|
||||
else
|
||||
delay <= {delay[N_STAGES-2:0], 1'b1};
|
||||
|
||||
assign rst_n_out = delay[N_STAGES-1];
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,39 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// A 2FF synchronizer to mitigate metastabilities. This is a baseline
|
||||
// implementation -- you should replace it with cells specific to your
|
||||
// FPGA/process
|
||||
|
||||
`ifndef HAZARD3_REG_KEEP_ATTRIBUTE
|
||||
`define HAZARD3_REG_KEEP_ATTRIBUTE (* keep = 1'b1 *) (* async_reg *)
|
||||
`endif
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_sync_1bit #(
|
||||
parameter N_STAGES = 2 // Should be >=2
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
input wire i,
|
||||
output wire o
|
||||
);
|
||||
|
||||
`HAZARD3_REG_KEEP_ATTRIBUTE reg [N_STAGES-1:0] sync_flops;
|
||||
|
||||
always @ (posedge clk or negedge rst_n)
|
||||
if (!rst_n)
|
||||
sync_flops <= {N_STAGES{1'b0}};
|
||||
else
|
||||
sync_flops <= {sync_flops[N_STAGES-2:0], i};
|
||||
|
||||
assign o = sync_flops[N_STAGES-1];
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+1
@@ -0,0 +1 @@
|
||||
file hazard3_dm.v
|
||||
+904
@@ -0,0 +1,904 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// RISC-V Debug Module for Hazard3. Supports up to 32 cores (1 hart per core).
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_dm #(
|
||||
// Where there are multiple harts per DM, the least-indexed hart is the
|
||||
// least-significant on each concatenated hart access bus.
|
||||
parameter N_HARTS = 1,
|
||||
// Where there are multiple DMs, the address of each DM should be a
|
||||
// multiple of 'h200, so that bits[8:2] decode correctly.
|
||||
parameter NEXT_DM_ADDR = 32'h0000_0000,
|
||||
// Implement support for system bus access:
|
||||
parameter HAVE_SBA = 0,
|
||||
|
||||
// Do not modify:
|
||||
parameter XLEN = 32, // Do not modify
|
||||
parameter W_HARTSEL = N_HARTS > 1 ? $clog2(N_HARTS) : 1 // Do not modify
|
||||
) (
|
||||
// DM is assumed to be in same clock domain as core; clock crossing
|
||||
// (if any) is inside DTM, or between DTM and DM.
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
|
||||
// APB access from Debug Transport Module
|
||||
input wire dmi_psel,
|
||||
input wire dmi_penable,
|
||||
input wire dmi_pwrite,
|
||||
input wire [8:0] dmi_paddr,
|
||||
input wire [31:0] dmi_pwdata,
|
||||
output reg [31:0] dmi_prdata,
|
||||
output wire dmi_pready,
|
||||
output wire dmi_pslverr,
|
||||
|
||||
// Reset request/acknowledge. "req" is a pulse >= 1 cycle wide. "done" is
|
||||
// level-sensitive, goes high once component is out of reset.
|
||||
//
|
||||
// The "sys" reset (ndmreset) is conventionally everything apart from DM +
|
||||
// DTM, but, as per section 3.2 in 0.13.2 debug spec: "Exactly what is
|
||||
// affected by this reset is implementation dependent, as long as it is
|
||||
// possible to debug programs from the first instruction executed." So
|
||||
// this could simply be an all-hart reset.
|
||||
output wire sys_reset_req,
|
||||
input wire sys_reset_done,
|
||||
output wire [N_HARTS-1:0] hart_reset_req,
|
||||
input wire [N_HARTS-1:0] hart_reset_done,
|
||||
|
||||
// Hart run/halt control
|
||||
output wire [N_HARTS-1:0] hart_req_halt,
|
||||
output wire [N_HARTS-1:0] hart_req_halt_on_reset,
|
||||
output wire [N_HARTS-1:0] hart_req_resume,
|
||||
input wire [N_HARTS-1:0] hart_halted,
|
||||
input wire [N_HARTS-1:0] hart_running,
|
||||
|
||||
// Hart access to data0 CSR (assumed to be core-internal but per-hart)
|
||||
output wire [N_HARTS*XLEN-1:0] hart_data0_rdata,
|
||||
input wire [N_HARTS*XLEN-1:0] hart_data0_wdata,
|
||||
input wire [N_HARTS-1:0] hart_data0_wen,
|
||||
|
||||
// Hart instruction injection
|
||||
output wire [N_HARTS*32-1:0] hart_instr_data,
|
||||
output reg [N_HARTS-1:0] hart_instr_data_vld,
|
||||
input wire [N_HARTS-1:0] hart_instr_data_rdy,
|
||||
input wire [N_HARTS-1:0] hart_instr_caught_exception,
|
||||
input wire [N_HARTS-1:0] hart_instr_caught_ebreak,
|
||||
|
||||
// System bus access (optional) -- can be hooked up to the standalone AHB
|
||||
// shim (hazard3_sbus_to_ahb.v) or the SBA input port on the processor
|
||||
// wrapper, which muxes SBA into the processor's load/store bus access
|
||||
// port. SBA does not increase debugger bus throughput, but supports
|
||||
// minimally intrusive debug bus access for e.g. Segger RTT.
|
||||
output wire [31:0] sbus_addr,
|
||||
output wire sbus_write,
|
||||
output wire [1:0] sbus_size,
|
||||
output wire sbus_vld,
|
||||
input wire sbus_rdy,
|
||||
input wire sbus_err,
|
||||
output wire [31:0] sbus_wdata,
|
||||
input wire [31:0] sbus_rdata
|
||||
);
|
||||
|
||||
wire dmi_write = dmi_psel && dmi_penable && dmi_pready && dmi_pwrite;
|
||||
wire dmi_read = dmi_psel && dmi_penable && dmi_pready && !dmi_pwrite;
|
||||
assign dmi_pready = 1'b1;
|
||||
assign dmi_pslverr = 1'b0;
|
||||
|
||||
// Program buffer is fixed at 2 words plus impebreak. The main thing we care
|
||||
// about is support for efficient memory block transfers using abstractauto;
|
||||
// in this case 2 words + impebreak is sufficient for RV32I, and 1 word +
|
||||
// impebreak is sufficient for RV32IC.
|
||||
localparam PROGBUF_SIZE = 2;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Address constants
|
||||
|
||||
localparam ADDR_DATA0 = 7'h04;
|
||||
// Other data registers not present.
|
||||
localparam ADDR_DMCONTROL = 7'h10;
|
||||
localparam ADDR_DMSTATUS = 7'h11;
|
||||
localparam ADDR_HARTINFO = 7'h12;
|
||||
localparam ADDR_HALTSUM1 = 7'h13;
|
||||
localparam ADDR_HALTSUM0 = 7'h40;
|
||||
// No HALTSUM2+ registers (we don't support >32 harts anyway)
|
||||
localparam ADDR_HAWINDOWSEL = 7'h14;
|
||||
localparam ADDR_HAWINDOW = 7'h15;
|
||||
localparam ADDR_ABSTRACTCS = 7'h16;
|
||||
localparam ADDR_COMMAND = 7'h17;
|
||||
localparam ADDR_ABSTRACTAUTO = 7'h18;
|
||||
localparam ADDR_CONFSTRPTR0 = 7'h19;
|
||||
localparam ADDR_CONFSTRPTR1 = 7'h1a;
|
||||
localparam ADDR_CONFSTRPTR2 = 7'h1b;
|
||||
localparam ADDR_CONFSTRPTR3 = 7'h1c;
|
||||
localparam ADDR_NEXTDM = 7'h1d;
|
||||
localparam ADDR_PROGBUF0 = 7'h20;
|
||||
localparam ADDR_PROGBUF1 = 7'h21;
|
||||
// No authentication
|
||||
localparam ADDR_SBCS = 7'h38;
|
||||
localparam ADDR_SBADDRESS0 = 7'h39;
|
||||
localparam ADDR_SBDATA0 = 7'h3c;
|
||||
|
||||
// APB is byte-addressed, DM registers are word-addressed.
|
||||
wire [6:0] dmi_regaddr = dmi_paddr[8:2];
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Hart selection
|
||||
|
||||
reg dmactive;
|
||||
|
||||
// Some fiddliness to make sure we get a single-wide zero-valued signal when
|
||||
// N_HARTS == 1 (so we can use this for indexing of per-hart signals)
|
||||
reg [W_HARTSEL-1:0] hartsel;
|
||||
wire [W_HARTSEL-1:0] hartsel_next;
|
||||
|
||||
generate
|
||||
if (N_HARTS > 1) begin: has_hartsel
|
||||
|
||||
// Only the lower 10 bits of hartsel are supported
|
||||
assign hartsel_next = dmi_write && dmi_regaddr == ADDR_DMCONTROL ?
|
||||
dmi_pwdata[16 +: W_HARTSEL] : hartsel;
|
||||
|
||||
end else begin: has_no_hartsel
|
||||
|
||||
assign hartsel_next = 1'b0;
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
hartsel <= {W_HARTSEL{1'b0}};
|
||||
end else if (!dmactive) begin
|
||||
hartsel <= {W_HARTSEL{1'b0}};
|
||||
end else begin
|
||||
hartsel <= hartsel_next;
|
||||
end
|
||||
end
|
||||
|
||||
// Also implement the hart array mask if there is more than one hart.
|
||||
reg [N_HARTS-1:0] hart_array_mask;
|
||||
reg hasel;
|
||||
wire [N_HARTS-1:0] hart_array_mask_next;
|
||||
wire hasel_next;
|
||||
|
||||
generate
|
||||
if (N_HARTS > 1) begin: has_array_mask
|
||||
|
||||
assign hart_array_mask_next = dmi_write && dmi_regaddr == ADDR_HAWINDOW ?
|
||||
dmi_pwdata[N_HARTS-1:0] : hart_array_mask;
|
||||
assign hasel_next = dmi_write && dmi_regaddr == ADDR_DMCONTROL ?
|
||||
dmi_pwdata[26] : hasel;
|
||||
|
||||
end else begin: has_no_array_mask
|
||||
|
||||
assign hart_array_mask_next = 1'b0;
|
||||
assign hasel_next = 1'b0;
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
hart_array_mask <= {N_HARTS{1'b0}};
|
||||
hasel <= 1'b0;
|
||||
end else if (!dmactive) begin
|
||||
hart_array_mask <= {N_HARTS{1'b0}};
|
||||
hasel <= 1'b0;
|
||||
end else begin
|
||||
hart_array_mask <= hart_array_mask_next;
|
||||
hasel <= hasel_next;
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Run/halt/reset control
|
||||
|
||||
// Normal read/write fields for dmcontrol (note some of these are per-hart
|
||||
// fields that get rotated into dmcontrol based on the current/next hartsel).
|
||||
reg [N_HARTS-1:0] dmcontrol_haltreq;
|
||||
reg [N_HARTS-1:0] dmcontrol_hartreset;
|
||||
reg [N_HARTS-1:0] dmcontrol_resethaltreq;
|
||||
reg dmcontrol_ndmreset;
|
||||
|
||||
wire [N_HARTS-1:0] dmcontrol_op_mask;
|
||||
|
||||
generate
|
||||
if (N_HARTS > 1) begin: dmcontrol_multiple_harts
|
||||
|
||||
// Selection is the hart selected by hartsel, *plus* the hart array mask
|
||||
// if hasel is set. Note we don't need to use the "next" version of
|
||||
// hart_array_mask since it can't change simultaneously with dmcontrol.
|
||||
assign dmcontrol_op_mask =
|
||||
(hartsel_next >= N_HARTS ? {N_HARTS{1'b0}} : {{N_HARTS-1{1'b0}}, 1'b1} << hartsel_next)
|
||||
| ({N_HARTS{hasel_next}} & hart_array_mask);
|
||||
|
||||
end else begin: dmcontrol_single_hart
|
||||
|
||||
assign dmcontrol_op_mask = 1'b1;
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
dmactive <= 1'b0;
|
||||
dmcontrol_ndmreset <= 1'b0;
|
||||
dmcontrol_haltreq <= {N_HARTS{1'b0}};
|
||||
dmcontrol_hartreset <= {N_HARTS{1'b0}};
|
||||
dmcontrol_resethaltreq <= {N_HARTS{1'b0}};
|
||||
end else if (!dmactive) begin
|
||||
// Only dmactive is writable when !dmactive
|
||||
if (dmi_write && dmi_regaddr == ADDR_DMCONTROL)
|
||||
dmactive <= dmi_pwdata[0];
|
||||
dmcontrol_ndmreset <= 1'b0;
|
||||
dmcontrol_haltreq <= {N_HARTS{1'b0}};
|
||||
dmcontrol_hartreset <= {N_HARTS{1'b0}};
|
||||
dmcontrol_resethaltreq <= {N_HARTS{1'b0}};
|
||||
end else if (dmi_write && dmi_regaddr == ADDR_DMCONTROL) begin
|
||||
dmactive <= dmi_pwdata[0];
|
||||
dmcontrol_ndmreset <= dmi_pwdata[1];
|
||||
|
||||
dmcontrol_haltreq <= (dmcontrol_haltreq & ~dmcontrol_op_mask) |
|
||||
({N_HARTS{dmi_pwdata[31]}} & dmcontrol_op_mask);
|
||||
|
||||
dmcontrol_hartreset <= (dmcontrol_hartreset & ~dmcontrol_op_mask) |
|
||||
({N_HARTS{dmi_pwdata[29]}} & dmcontrol_op_mask);
|
||||
|
||||
dmcontrol_resethaltreq <= (dmcontrol_resethaltreq
|
||||
& ~({N_HARTS{dmi_pwdata[2]}} & dmcontrol_op_mask))
|
||||
| ({N_HARTS{dmi_pwdata[3]}} & dmcontrol_op_mask);
|
||||
end
|
||||
end
|
||||
|
||||
assign sys_reset_req = dmcontrol_ndmreset;
|
||||
assign hart_reset_req = dmcontrol_hartreset;
|
||||
assign hart_req_halt = dmcontrol_haltreq;
|
||||
assign hart_req_halt_on_reset = dmcontrol_resethaltreq;
|
||||
|
||||
reg [N_HARTS-1:0] hart_reset_done_prev;
|
||||
reg [N_HARTS-1:0] dmstatus_havereset;
|
||||
wire [N_HARTS-1:0] hart_available = hart_reset_done & {N_HARTS{sys_reset_done}};
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
hart_reset_done_prev <= {N_HARTS{1'b0}};
|
||||
end else begin
|
||||
hart_reset_done_prev <= hart_reset_done;
|
||||
end
|
||||
end
|
||||
|
||||
wire dmcontrol_ackhavereset = dmi_write && dmi_regaddr == ADDR_DMCONTROL && dmi_pwdata[28];
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
dmstatus_havereset <= {N_HARTS{1'b0}};
|
||||
end else if (!dmactive) begin
|
||||
dmstatus_havereset <= {N_HARTS{1'b0}};
|
||||
end else begin
|
||||
dmstatus_havereset <= (dmstatus_havereset | (hart_reset_done & ~hart_reset_done_prev))
|
||||
& ~({N_HARTS{dmcontrol_ackhavereset}} & dmcontrol_op_mask);
|
||||
end
|
||||
end
|
||||
|
||||
reg [N_HARTS-1:0] dmstatus_resumeack;
|
||||
reg [N_HARTS-1:0] dmcontrol_resumereq_sticky;
|
||||
|
||||
// Note: we are required to ignore resumereq when haltreq is also set, as per
|
||||
// spec (odd since the host is forbidden from writing both at once anyway).
|
||||
// The wording is odd, it refers only to `haltreq` which is specifically the
|
||||
// write-only `dmcontrol` field, not the underlying halt request state bits.
|
||||
wire dmcontrol_resumereq = dmi_write && dmi_regaddr == ADDR_DMCONTROL &&
|
||||
dmi_pwdata[30] && !dmi_pwdata[31];
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
dmstatus_resumeack <= {N_HARTS{1'b0}};
|
||||
dmcontrol_resumereq_sticky <= {N_HARTS{1'b0}};
|
||||
end else if (!dmactive) begin
|
||||
dmstatus_resumeack <= {N_HARTS{1'b0}};
|
||||
dmcontrol_resumereq_sticky <= {N_HARTS{1'b0}};
|
||||
end else begin
|
||||
dmstatus_resumeack <= (dmstatus_resumeack
|
||||
| (dmcontrol_resumereq_sticky & hart_running & hart_available))
|
||||
& ~({N_HARTS{dmcontrol_resumereq}} & dmcontrol_op_mask);
|
||||
|
||||
dmcontrol_resumereq_sticky <= (dmcontrol_resumereq_sticky
|
||||
& ~(hart_running & hart_available))
|
||||
| ({N_HARTS{dmcontrol_resumereq}} & dmcontrol_op_mask);
|
||||
end
|
||||
end
|
||||
|
||||
assign hart_req_resume = dmcontrol_resumereq_sticky;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// System bus access
|
||||
|
||||
reg [31:0] sbaddress;
|
||||
reg [31:0] sbdata;
|
||||
|
||||
// Update logic for address/data registers:
|
||||
|
||||
reg sbbusy;
|
||||
reg sbautoincrement;
|
||||
reg [2:0] sbaccess; // Size of the transfer
|
||||
|
||||
wire sbdata_write_blocked;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
sbaddress <= {32{1'b0}};
|
||||
sbdata <= {32{1'b0}};
|
||||
end else if (!dmactive) begin
|
||||
sbaddress <= {32{1'b0}};
|
||||
sbdata <= {32{1'b0}};
|
||||
end else if (HAVE_SBA) begin
|
||||
if (dmi_write && dmi_regaddr == ADDR_SBDATA0 && !sbdata_write_blocked) begin
|
||||
// Note sbbusyerror and sberror block writes to sbdata0, as the
|
||||
// write is required to have no side effects when they are set.
|
||||
sbdata <= dmi_pwdata;
|
||||
end else if (sbus_vld && sbus_rdy && !sbus_write && !sbus_err) begin
|
||||
// Make sure the lower byte lanes see appropriately shifted data as
|
||||
// long as the transfer is naturally aligned
|
||||
sbdata <= sbaddress[1:0] == 2'b01 ? {sbus_rdata[31:8], sbus_rdata[15:8]} :
|
||||
sbaddress[1:0] == 2'b10 ? {sbus_rdata[31:16], sbus_rdata[31:16]} :
|
||||
sbaddress[1:0] == 2'b11 ? {sbus_rdata[31:8], sbus_rdata[31:24]} : sbus_rdata;
|
||||
end
|
||||
if (dmi_write && dmi_regaddr == ADDR_SBADDRESS0 && !sbbusy) begin
|
||||
// Note sbaddress can't be written when busy, but
|
||||
// sberror/sbbusyerror do not prevent writes.
|
||||
sbaddress <= dmi_pwdata;
|
||||
end else if (sbus_vld && sbus_rdy && !sbus_err && sbautoincrement) begin
|
||||
// Note: address increments only following a successful transfer.
|
||||
// Spec 0.13.2 weirdly implies address should increment following
|
||||
// a sbdata0 read with sbautoincrement=1 and sbreadondata=0, but
|
||||
// this seems to be a typo, fixed in later versions.
|
||||
sbaddress <= sbaddress + (
|
||||
sbaccess[1:0] == 2'b00 ? 32'd1 :
|
||||
sbaccess[1:0] == 2'b01 ? 32'd2 : 32'd4
|
||||
);
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// Control logic:
|
||||
|
||||
reg sbbusyerror;
|
||||
reg sbreadonaddr;
|
||||
reg sbreadondata;
|
||||
reg [2:0] sberror;
|
||||
reg sb_current_is_write;
|
||||
|
||||
localparam SBERROR_OK = 3'h0;
|
||||
localparam SBERROR_BADADDR = 3'h2;
|
||||
localparam SBERROR_BADALIGN = 3'h3;
|
||||
localparam SBERROR_BADSIZE = 3'h4;
|
||||
|
||||
assign sbdata_write_blocked = sbbusy || sbbusyerror || |sberror;
|
||||
|
||||
// Notes on behaviour of sbbusyerror: the sbbusyerror description says:
|
||||
//
|
||||
// "Set when the debugger attempts to read data while a read is in progress,
|
||||
// or when the debugger initiates a new access while one is already in
|
||||
// progress (while sbbusy is set)."
|
||||
//
|
||||
// However, sbaddress0 description says:
|
||||
//
|
||||
// "When the system bus master is busy, writes to this register will set
|
||||
// sbbusyerror and don’t do anything else."
|
||||
//
|
||||
// ...not conditioned on sbreadonaddr. Likewise the sbdata0 description says:
|
||||
//
|
||||
// "If the bus master is busy then accesses set sbbusyerror, and don’t do
|
||||
// anything else."
|
||||
//
|
||||
// ...not conditioned on sbreadondata. We are going to take the union of all
|
||||
// the cases where the spec says we should raise an error:
|
||||
|
||||
wire sb_access_illegal_when_busy =
|
||||
dmi_regaddr == ADDR_SBDATA0 && (dmi_read || dmi_write) ||
|
||||
dmi_regaddr == ADDR_SBADDRESS0 && dmi_write;
|
||||
|
||||
wire sb_want_start_write = dmi_write && dmi_regaddr == ADDR_SBDATA0;
|
||||
|
||||
wire sb_want_start_read =
|
||||
(sbreadonaddr && dmi_write && dmi_regaddr == ADDR_SBADDRESS0) ||
|
||||
(sbreadondata && dmi_read && dmi_regaddr == ADDR_SBDATA0);
|
||||
|
||||
wire [1:0] sb_next_align = sbreadonaddr && dmi_write && dmi_regaddr == ADDR_SBADDRESS0 ?
|
||||
dmi_pwdata[1:0] : sbaddress[1:0];
|
||||
|
||||
wire sb_badalign =
|
||||
(sbaccess == 3'h1 && sb_next_align[0]) ||
|
||||
(sbaccess == 3'h2 && |sb_next_align[1:0]);
|
||||
|
||||
wire sb_badsize = sbaccess > 3'h2;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
sbbusy <= 1'b0;
|
||||
sbbusyerror <= 1'b0;
|
||||
sbreadonaddr <= 1'b0;
|
||||
sbreadondata <= 1'b0;
|
||||
sbaccess <= 3'h0;
|
||||
sbautoincrement <= 1'b0;
|
||||
sberror <= 3'h0;
|
||||
sb_current_is_write <= 1'b0;
|
||||
end else if (!dmactive) begin
|
||||
sbbusy <= 1'b0;
|
||||
sbbusyerror <= 1'b0;
|
||||
sbreadonaddr <= 1'b0;
|
||||
sbreadondata <= 1'b0;
|
||||
sbaccess <= 3'h0;
|
||||
sbautoincrement <= 1'b0;
|
||||
sberror <= 3'h0;
|
||||
sb_current_is_write <= 1'b0;
|
||||
end else if (HAVE_SBA) begin
|
||||
if (dmi_write && dmi_regaddr == ADDR_SBCS) begin
|
||||
// Assume a transfer is not in progress when written (per spec)
|
||||
sbbusyerror <= sbbusyerror && !dmi_pwdata[22];
|
||||
sbreadonaddr <= dmi_pwdata[20];
|
||||
sbaccess <= dmi_pwdata[19:17];
|
||||
sbautoincrement <= dmi_pwdata[16];
|
||||
sbreadondata <= dmi_pwdata[15];
|
||||
sberror <= sberror & ~dmi_pwdata[14:12];
|
||||
end
|
||||
if (sbbusy) begin
|
||||
if (sb_access_illegal_when_busy) begin
|
||||
sbbusyerror <= 1'b1;
|
||||
end
|
||||
if (sbus_vld && sbus_rdy) begin
|
||||
sbbusy <= 1'b0;
|
||||
if (sbus_err) begin
|
||||
sberror <= SBERROR_BADADDR;
|
||||
end
|
||||
end
|
||||
end else if ((sb_want_start_read || sb_want_start_write) && ~|sberror && !sbbusyerror) begin
|
||||
if (sb_badsize) begin
|
||||
sberror <= SBERROR_BADSIZE;
|
||||
end else if (sb_badalign) begin
|
||||
sberror <= SBERROR_BADALIGN;
|
||||
end else begin
|
||||
sbbusy <= 1'b1;
|
||||
sb_current_is_write <= sb_want_start_write;
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
assign sbus_addr = sbaddress;
|
||||
assign sbus_write = sb_current_is_write;
|
||||
assign sbus_size = sbaccess[1:0];
|
||||
assign sbus_vld = sbbusy;
|
||||
|
||||
// Replicate byte lanes to handle naturally-aligned cases.
|
||||
assign sbus_wdata = sbaccess[1:0] == 2'b00 ? {4{sbdata[7:0]}} :
|
||||
sbaccess[1:0] == 2'b01 ? {2{sbdata[15:0]}} : sbdata;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Abstract command data registers
|
||||
|
||||
wire abstractcs_busy;
|
||||
|
||||
// The same data0 register is aliased as a CSR on all harts connected to this
|
||||
// DM. Cores may read data0 as a CSR when in debug mode, and may write it when:
|
||||
//
|
||||
// - That core is in debug mode, and...
|
||||
// - We are currently executing an abstract command on that core
|
||||
//
|
||||
// The DM can also read/write data0 at all times.
|
||||
|
||||
reg [XLEN-1:0] abstract_data0;
|
||||
|
||||
assign hart_data0_rdata = {N_HARTS{abstract_data0}};
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin: update_hart_data0
|
||||
reg signed [31:0] i;
|
||||
if (!rst_n) begin
|
||||
abstract_data0 <= {XLEN{1'b0}};
|
||||
end else if (!dmactive) begin
|
||||
abstract_data0 <= {XLEN{1'b0}};
|
||||
end else if (dmi_write && dmi_regaddr == ADDR_DATA0) begin
|
||||
abstract_data0 <= dmi_pwdata;
|
||||
end else begin
|
||||
for (i = 0; i < N_HARTS; i = i + 1) begin
|
||||
if (hartsel == i[W_HARTSEL-1:0] && hart_data0_wen[i] && hart_halted[i] && abstractcs_busy)
|
||||
abstract_data0 <= hart_data0_wdata[i * XLEN +: XLEN];
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
reg [XLEN-1:0] progbuf0;
|
||||
reg [XLEN-1:0] progbuf1;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
progbuf0 <= {XLEN{1'b0}};
|
||||
progbuf1 <= {XLEN{1'b0}};
|
||||
end else if (!dmactive) begin
|
||||
progbuf0 <= {XLEN{1'b0}};
|
||||
progbuf1 <= {XLEN{1'b0}};
|
||||
end else if (dmi_write && !abstractcs_busy) begin
|
||||
if (dmi_regaddr == ADDR_PROGBUF0)
|
||||
progbuf0 <= dmi_pwdata;
|
||||
if (dmi_regaddr == ADDR_PROGBUF1)
|
||||
progbuf1 <= dmi_pwdata;
|
||||
end
|
||||
end
|
||||
|
||||
reg abstractauto_autoexecdata;
|
||||
reg [1:0] abstractauto_autoexecprogbuf;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
abstractauto_autoexecdata <= 1'b0;
|
||||
abstractauto_autoexecprogbuf <= 2'b00;
|
||||
end else if (!dmactive) begin
|
||||
abstractauto_autoexecdata <= 1'b0;
|
||||
abstractauto_autoexecprogbuf <= 2'b00;
|
||||
end else if (dmi_write && dmi_regaddr == ADDR_ABSTRACTAUTO) begin
|
||||
abstractauto_autoexecdata <= dmi_pwdata[0];
|
||||
abstractauto_autoexecprogbuf <= dmi_pwdata[17:16];
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Abstract command state machine
|
||||
|
||||
localparam W_STATE = 4;
|
||||
localparam S_IDLE = 4'd0;
|
||||
|
||||
localparam S_ISSUE_REGREAD = 4'd1;
|
||||
localparam S_ISSUE_REGWRITE = 4'd2;
|
||||
localparam S_ISSUE_REGEBREAK = 4'd3;
|
||||
localparam S_WAIT_REGEBREAK = 4'd4;
|
||||
|
||||
localparam S_ISSUE_PROGBUF0 = 4'd5;
|
||||
localparam S_ISSUE_PROGBUF1 = 4'd6;
|
||||
localparam S_ISSUE_IMPEBREAK = 4'd7;
|
||||
localparam S_WAIT_IMPEBREAK = 4'd8;
|
||||
|
||||
localparam CMDERR_OK = 3'h0;
|
||||
localparam CMDERR_BUSY = 3'h1;
|
||||
localparam CMDERR_UNSUPPORTED = 3'h2;
|
||||
localparam CMDERR_EXCEPTION = 3'h3;
|
||||
localparam CMDERR_HALTRESUME = 3'h4;
|
||||
|
||||
reg [2:0] abstractcs_cmderr;
|
||||
reg [2:0] abstractcs_cmderr_nxt;
|
||||
reg [W_STATE-1:0] acmd_state;
|
||||
reg [W_STATE-1:0] acmd_state_nxt;
|
||||
|
||||
assign abstractcs_busy = acmd_state != S_IDLE;
|
||||
|
||||
wire start_abstract_cmd = abstractcs_cmderr == CMDERR_OK && !abstractcs_busy && (
|
||||
(dmi_write && dmi_regaddr == ADDR_COMMAND) ||
|
||||
((dmi_write || dmi_read) && abstractauto_autoexecdata && dmi_regaddr == ADDR_DATA0) ||
|
||||
((dmi_write || dmi_read) && abstractauto_autoexecprogbuf[0] && dmi_regaddr == ADDR_PROGBUF0) ||
|
||||
((dmi_write || dmi_read) && abstractauto_autoexecprogbuf[1] && dmi_regaddr == ADDR_PROGBUF1)
|
||||
);
|
||||
|
||||
wire dmi_access_illegal_when_busy =
|
||||
(dmi_write && (
|
||||
dmi_regaddr == ADDR_ABSTRACTCS || dmi_regaddr == ADDR_COMMAND || dmi_regaddr == ADDR_ABSTRACTAUTO ||
|
||||
dmi_regaddr == ADDR_DATA0 || dmi_regaddr == ADDR_PROGBUF0 || dmi_regaddr == ADDR_PROGBUF1)) ||
|
||||
(dmi_read && (
|
||||
dmi_regaddr == ADDR_DATA0 || dmi_regaddr == ADDR_PROGBUF0 || dmi_regaddr == ADDR_PROGBUF1));
|
||||
|
||||
// Decode what acmd may be triggered on this cycle, and whether it is
|
||||
// supported -- command source may be a registered version of most recent
|
||||
// command (if abstractauto is used) or a fresh command off the bus. We don't
|
||||
// register the entire write data; repeats of unsupported commands are
|
||||
// detected by just registering that the last written command was
|
||||
// unsupported.
|
||||
|
||||
wire acmd_new = dmi_write && dmi_regaddr == ADDR_COMMAND;
|
||||
|
||||
wire acmd_new_postexec = dmi_pwdata[18];
|
||||
wire acmd_new_transfer = dmi_pwdata[17];
|
||||
wire acmd_new_write = dmi_pwdata[16];
|
||||
wire [4:0] acmd_new_regno = dmi_pwdata[4:0];
|
||||
|
||||
// Note: regno and aarsize are permitted to have otherwise-invalid values if
|
||||
// the transfer flag is not set.
|
||||
wire acmd_new_unsupported =
|
||||
(dmi_pwdata[31:24] != 8'h00 ) || // Only Access Register command supported
|
||||
(dmi_pwdata[22:20] != 3'h2 && acmd_new_transfer) || // Must be 32 bits in size
|
||||
(dmi_pwdata[19] ) || // aarpostincrement not supported
|
||||
(dmi_pwdata[15:12] != 4'h1 && acmd_new_transfer) || // Only core register access supported
|
||||
(dmi_pwdata[11:5] != 7'h0 && acmd_new_transfer); // Only GPRs supported
|
||||
|
||||
reg acmd_prev_postexec;
|
||||
reg acmd_prev_transfer;
|
||||
reg acmd_prev_write;
|
||||
reg [4:0] acmd_prev_regno;
|
||||
reg acmd_prev_unsupported;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
acmd_prev_postexec <= 1'b0;
|
||||
acmd_prev_transfer <= 1'b0;
|
||||
acmd_prev_write <= 1'b0;
|
||||
acmd_prev_regno <= 5'h0;
|
||||
acmd_prev_unsupported <= 1'b1;
|
||||
end else if (!dmactive) begin
|
||||
acmd_prev_postexec <= 1'b0;
|
||||
acmd_prev_transfer <= 1'b0;
|
||||
acmd_prev_write <= 1'b0;
|
||||
acmd_prev_regno <= 5'h0;
|
||||
acmd_prev_unsupported <= 1'b1;
|
||||
end else if (start_abstract_cmd && acmd_new) begin
|
||||
acmd_prev_postexec <= acmd_new_postexec;
|
||||
acmd_prev_transfer <= acmd_new_transfer;
|
||||
acmd_prev_write <= acmd_new_write;
|
||||
acmd_prev_regno <= acmd_new_regno;
|
||||
acmd_prev_unsupported <= acmd_new_unsupported;
|
||||
end
|
||||
end
|
||||
|
||||
wire acmd_postexec = acmd_new ? acmd_new_postexec : acmd_prev_postexec ;
|
||||
wire acmd_transfer = acmd_new ? acmd_new_transfer : acmd_prev_transfer ;
|
||||
wire acmd_write = acmd_new ? acmd_new_write : acmd_prev_write ;
|
||||
wire [4:0] acmd_regno = acmd_new ? acmd_new_regno : acmd_prev_regno ;
|
||||
wire acmd_unsupported = acmd_new ? acmd_new_unsupported : acmd_prev_unsupported;
|
||||
|
||||
always @ (*) begin
|
||||
// Default: no state change
|
||||
acmd_state_nxt = acmd_state;
|
||||
abstractcs_cmderr_nxt = abstractcs_cmderr;
|
||||
|
||||
if (dmi_write && dmi_regaddr == ADDR_ABSTRACTCS && !abstractcs_busy)
|
||||
abstractcs_cmderr_nxt = abstractcs_cmderr & ~dmi_pwdata[10:8];
|
||||
if (abstractcs_cmderr == CMDERR_OK && abstractcs_busy && dmi_access_illegal_when_busy)
|
||||
abstractcs_cmderr_nxt = CMDERR_BUSY;
|
||||
if (acmd_state != S_IDLE && hart_instr_caught_exception[hartsel])
|
||||
abstractcs_cmderr_nxt = CMDERR_EXCEPTION;
|
||||
|
||||
case (acmd_state)
|
||||
S_IDLE: begin
|
||||
if (start_abstract_cmd) begin
|
||||
if (!hart_halted[hartsel] || !hart_available[hartsel]) begin
|
||||
abstractcs_cmderr_nxt = CMDERR_HALTRESUME;
|
||||
end else if (acmd_unsupported) begin
|
||||
abstractcs_cmderr_nxt = CMDERR_UNSUPPORTED;
|
||||
end else begin
|
||||
if (acmd_transfer && acmd_write)
|
||||
acmd_state_nxt = S_ISSUE_REGWRITE;
|
||||
else if (acmd_transfer && !acmd_write)
|
||||
acmd_state_nxt = S_ISSUE_REGREAD;
|
||||
else if (acmd_postexec)
|
||||
acmd_state_nxt = S_ISSUE_PROGBUF0;
|
||||
else
|
||||
acmd_state_nxt = S_IDLE;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
S_ISSUE_REGREAD: begin
|
||||
if (hart_instr_data_rdy[hartsel])
|
||||
acmd_state_nxt = S_ISSUE_REGEBREAK;
|
||||
end
|
||||
S_ISSUE_REGWRITE: begin
|
||||
if (hart_instr_data_rdy[hartsel])
|
||||
acmd_state_nxt = S_ISSUE_REGEBREAK;
|
||||
end
|
||||
S_ISSUE_REGEBREAK: begin
|
||||
if (hart_instr_data_rdy[hartsel])
|
||||
acmd_state_nxt = S_WAIT_REGEBREAK;
|
||||
end
|
||||
S_WAIT_REGEBREAK: begin
|
||||
if (hart_instr_caught_ebreak[hartsel]) begin
|
||||
if (acmd_prev_postexec)
|
||||
acmd_state_nxt = S_ISSUE_PROGBUF0;
|
||||
else
|
||||
acmd_state_nxt = S_IDLE;
|
||||
end
|
||||
end
|
||||
|
||||
S_ISSUE_PROGBUF0: begin
|
||||
if (hart_instr_data_rdy[hartsel])
|
||||
acmd_state_nxt = S_ISSUE_PROGBUF1;
|
||||
end
|
||||
S_ISSUE_PROGBUF1: begin
|
||||
if (hart_instr_caught_exception[hartsel] || hart_instr_caught_ebreak[hartsel]) begin
|
||||
acmd_state_nxt = S_IDLE;
|
||||
end else if (hart_instr_data_rdy[hartsel]) begin
|
||||
acmd_state_nxt = S_ISSUE_IMPEBREAK;
|
||||
end
|
||||
end
|
||||
S_ISSUE_IMPEBREAK: begin
|
||||
if (hart_instr_caught_exception[hartsel] || hart_instr_caught_ebreak[hartsel]) begin
|
||||
acmd_state_nxt = S_IDLE;
|
||||
end else if (hart_instr_data_rdy[hartsel]) begin
|
||||
acmd_state_nxt = S_WAIT_IMPEBREAK;
|
||||
end
|
||||
end
|
||||
S_WAIT_IMPEBREAK: begin
|
||||
if (hart_instr_caught_exception[hartsel] || hart_instr_caught_ebreak[hartsel]) begin
|
||||
acmd_state_nxt = S_IDLE;
|
||||
end
|
||||
end
|
||||
|
||||
default: begin
|
||||
// Unreachable
|
||||
end
|
||||
|
||||
endcase
|
||||
end
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
abstractcs_cmderr <= CMDERR_OK;
|
||||
acmd_state <= S_IDLE;
|
||||
end else if (!dmactive) begin
|
||||
abstractcs_cmderr <= CMDERR_OK;
|
||||
acmd_state <= S_IDLE;
|
||||
end else begin
|
||||
abstractcs_cmderr <= abstractcs_cmderr_nxt;
|
||||
acmd_state <= acmd_state_nxt;
|
||||
end
|
||||
end
|
||||
|
||||
wire [N_HARTS-1:0] hart_instr_data_vld_nxt = {{N_HARTS-1{1'b0}},
|
||||
acmd_state_nxt == S_ISSUE_REGREAD || acmd_state_nxt == S_ISSUE_REGWRITE || acmd_state_nxt == S_ISSUE_REGEBREAK ||
|
||||
acmd_state_nxt == S_ISSUE_PROGBUF0 || acmd_state_nxt == S_ISSUE_PROGBUF1 || acmd_state_nxt == S_ISSUE_IMPEBREAK
|
||||
} << hartsel;
|
||||
|
||||
wire [31:0] hart_instr_data_nxt =
|
||||
acmd_state_nxt == S_ISSUE_REGWRITE ? 32'hbff02073 | {20'd0, acmd_regno, 7'd0} : // csrr xx, dmdata0
|
||||
acmd_state_nxt == S_ISSUE_REGREAD ? 32'hbff01073 | {12'd0, acmd_regno, 15'd0} : // csrw dmdata0, xx
|
||||
acmd_state_nxt == S_ISSUE_PROGBUF0 ? progbuf0 :
|
||||
acmd_state_nxt == S_ISSUE_PROGBUF1 ? progbuf1 :
|
||||
32'h00100073; // ebreak
|
||||
|
||||
reg [31:0] hart_instr_data_reg;
|
||||
assign hart_instr_data = {N_HARTS{hart_instr_data_reg}};
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
hart_instr_data_vld <= 1'b0;
|
||||
hart_instr_data_reg <= 32'h00000000;
|
||||
end else begin
|
||||
hart_instr_data_vld <= hart_instr_data_vld_nxt;
|
||||
if (hart_instr_data_vld_nxt) begin
|
||||
hart_instr_data_reg <= hart_instr_data_nxt;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Status helper functions
|
||||
|
||||
function status_any;
|
||||
input [N_HARTS-1:0] status_mask;
|
||||
begin
|
||||
status_any = status_mask[hartsel] || (hasel && |(status_mask & hart_array_mask));
|
||||
end
|
||||
endfunction
|
||||
|
||||
function status_all;
|
||||
input [N_HARTS-1:0] status_mask;
|
||||
begin
|
||||
status_all = status_mask[hartsel] && (!hasel || ~|(~status_mask & hart_array_mask));
|
||||
end
|
||||
endfunction
|
||||
|
||||
function [1:0] status_all_any;
|
||||
input [N_HARTS-1:0] status_mask;
|
||||
begin
|
||||
status_all_any = {
|
||||
status_all(status_mask),
|
||||
status_any(status_mask)
|
||||
};
|
||||
end
|
||||
endfunction
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// DMI read data mux
|
||||
|
||||
always @ (*) begin
|
||||
case (dmi_regaddr)
|
||||
ADDR_DATA0: dmi_prdata = abstract_data0;
|
||||
ADDR_DMCONTROL: dmi_prdata = {
|
||||
1'b0, // haltreq is a W-only field
|
||||
1'b0, // resumereq is a W1 field
|
||||
status_any(dmcontrol_hartreset),
|
||||
1'b0, // ackhavereset is a W1 field
|
||||
1'b0, // reserved
|
||||
hasel,
|
||||
{{10-W_HARTSEL{1'b0}}, hartsel}, // hartsello
|
||||
10'h0, // hartselhi
|
||||
2'h0, // reserved
|
||||
2'h0, // set/clrresethaltreq are W1 fields
|
||||
dmcontrol_ndmreset,
|
||||
dmactive
|
||||
};
|
||||
ADDR_DMSTATUS: dmi_prdata = {
|
||||
9'h0, // reserved
|
||||
1'b1, // impebreak = 1
|
||||
2'h0, // reserved
|
||||
status_all_any(dmstatus_havereset), // allhavereset, anyhavereset
|
||||
status_all_any(dmstatus_resumeack), // allresumeack, anyresumeack
|
||||
hartsel >= N_HARTS && !(hasel && |hart_array_mask), // allnonexistent
|
||||
hartsel >= N_HARTS, // anynonexistent
|
||||
status_all_any(~hart_available), // allunavail, anyunavail
|
||||
status_all_any(hart_running & hart_available), // allrunning, anyrunning
|
||||
status_all_any(hart_halted & hart_available), // allhalted, anyhalted
|
||||
1'b1, // authenticated
|
||||
1'b0, // authbusy
|
||||
1'b1, // hasresethaltreq = 1 (we do support it)
|
||||
1'b0, // confstrptrvalid
|
||||
4'd2 // version = 2: RISC-V debug spec 0.13.2
|
||||
};
|
||||
ADDR_HARTINFO: dmi_prdata = {
|
||||
8'h0, // reserved
|
||||
4'h0, // nscratch = 0
|
||||
3'h0, // reserved
|
||||
1'b0, // dataccess = 0, data0 is mapped to each hart's CSR space
|
||||
4'h1, // datasize = 1, a single data CSR (data0) is available
|
||||
12'hbff // dataaddr, placed at the top of the M-custom space since
|
||||
// the spec doesn't reserve a location for it.
|
||||
};
|
||||
ADDR_HALTSUM0: dmi_prdata = {
|
||||
{XLEN - N_HARTS{1'b0}},
|
||||
hart_halted & hart_available
|
||||
};
|
||||
ADDR_HALTSUM1: dmi_prdata = {
|
||||
{XLEN - 1{1'b0}},
|
||||
|(hart_halted & hart_available)
|
||||
};
|
||||
ADDR_HAWINDOWSEL: dmi_prdata = 32'h00000000;
|
||||
ADDR_HAWINDOW: dmi_prdata = {
|
||||
{32-N_HARTS{1'b0}},
|
||||
hart_array_mask
|
||||
};
|
||||
ADDR_ABSTRACTCS: dmi_prdata = {
|
||||
3'h0, // reserved
|
||||
5'd2, // progbufsize = 2
|
||||
11'h0, // reserved
|
||||
abstractcs_busy,
|
||||
1'b0,
|
||||
abstractcs_cmderr,
|
||||
4'h0,
|
||||
4'd1 // datacount = 1
|
||||
};
|
||||
ADDR_ABSTRACTAUTO: dmi_prdata = {
|
||||
14'h0,
|
||||
abstractauto_autoexecprogbuf, // only progbuf0,1 present
|
||||
15'h0,
|
||||
abstractauto_autoexecdata // only data0 present
|
||||
};
|
||||
ADDR_SBCS: dmi_prdata = {
|
||||
3'h1, // version = 1
|
||||
6'h00,
|
||||
sbbusyerror,
|
||||
sbbusy,
|
||||
sbreadonaddr,
|
||||
sbaccess,
|
||||
sbautoincrement,
|
||||
sbreadondata,
|
||||
sberror,
|
||||
7'h20, // sbasize = 32
|
||||
5'b00111 // 8, 16, 32-bit transfers supported
|
||||
} & {32{|HAVE_SBA}};
|
||||
ADDR_SBDATA0: dmi_prdata = sbdata & {32{|HAVE_SBA}};
|
||||
ADDR_SBADDRESS0: dmi_prdata = sbaddress & {32{|HAVE_SBA}};
|
||||
ADDR_CONFSTRPTR0: dmi_prdata = 32'h4c296328;
|
||||
ADDR_CONFSTRPTR1: dmi_prdata = 32'h20656b75;
|
||||
ADDR_CONFSTRPTR2: dmi_prdata = 32'h6e657257;
|
||||
ADDR_CONFSTRPTR3: dmi_prdata = 32'h31322720;
|
||||
ADDR_NEXTDM: dmi_prdata = NEXT_DM_ADDR;
|
||||
ADDR_PROGBUF0: dmi_prdata = progbuf0;
|
||||
ADDR_PROGBUF1: dmi_prdata = progbuf1;
|
||||
default: dmi_prdata = {XLEN{1'b0}};
|
||||
endcase
|
||||
end
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,74 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Standalone bus shim for connecting the DM's System Bus Access to AHB
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_sbus_to_ahb #(
|
||||
parameter W_ADDR = 32,
|
||||
parameter W_DATA = 32
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
|
||||
input wire [W_ADDR-1:0] sbus_addr,
|
||||
input wire sbus_write,
|
||||
input wire [1:0] sbus_size,
|
||||
input wire sbus_vld,
|
||||
output wire sbus_rdy,
|
||||
output wire sbus_err,
|
||||
input wire [W_DATA-1:0] sbus_wdata,
|
||||
output wire [W_DATA-1:0] sbus_rdata,
|
||||
|
||||
output wire [W_ADDR-1:0] ahblm_haddr,
|
||||
output wire ahblm_hwrite,
|
||||
output wire [1:0] ahblm_htrans,
|
||||
output wire [2:0] ahblm_hsize,
|
||||
output wire [2:0] ahblm_hburst,
|
||||
output wire [3:0] ahblm_hprot,
|
||||
output wire ahblm_hmastlock,
|
||||
input wire ahblm_hready,
|
||||
input wire ahblm_hresp,
|
||||
output wire [W_DATA-1:0] ahblm_hwdata,
|
||||
input wire [W_DATA-1:0] ahblm_hrdata
|
||||
);
|
||||
|
||||
// Most signals are simple tie-throughs
|
||||
|
||||
assign ahblm_haddr = sbus_addr;
|
||||
assign ahblm_hwrite = sbus_write;
|
||||
assign ahblm_hsize = {1'b0, sbus_size};
|
||||
assign ahblm_hwdata = sbus_wdata;
|
||||
|
||||
// HPROT = noncacheable nonbufferable privileged data access:
|
||||
assign ahblm_hprot = 4'b0011;
|
||||
assign ahblm_hmastlock = 1'b0;
|
||||
assign ahblm_hburst = 3'h0;
|
||||
|
||||
assign sbus_err = ahblm_hresp;
|
||||
assign sbus_rdata = ahblm_hrdata;
|
||||
|
||||
// Handshaking
|
||||
|
||||
reg dph_active;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
dph_active <= 1'b0;
|
||||
end else if (ahblm_hready) begin
|
||||
dph_active <= ahblm_htrans[1];
|
||||
end
|
||||
end
|
||||
|
||||
assign ahblm_htrans = sbus_vld && !dph_active ? 2'b10 : 2'b00;
|
||||
|
||||
assign sbus_rdy = ahblm_hready && dph_active;
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,6 @@
|
||||
file hazard3_ecp5_jtag_dtm.v
|
||||
file hazard3_jtag_dtm_core.v
|
||||
|
||||
file ../cdc/hazard3_apb_async_bridge.v
|
||||
file ../cdc/hazard3_reset_sync.v
|
||||
file ../cdc/hazard3_sync_1bit.v
|
||||
@@ -0,0 +1,194 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// The ECP5 JTAGG primitive (yes that is the correct spelling) allows you to
|
||||
// add two custom DRs to the FPGA's chip TAP, selected using the 8-bit ER1
|
||||
// (0x32) and ER2 (0x38) instructions.
|
||||
//
|
||||
// Brian Swetland pointed out on Twitter that the standard RISC-V JTAG-DTM
|
||||
// only uses two DRs (DTMCS and DMI), besides the standard IDCODE and BYPASS
|
||||
// which are provided already by the ECP5 TAP. This file instantiates the
|
||||
// guts of Hazard3's standard JTAG-DTM and connects the DTMCS and DMI
|
||||
// registers to the JTAGG primitive's ER1/ER2 DRs.
|
||||
//
|
||||
// The exciting part is that upstream OpenOCD already allows you to set the IR
|
||||
// length *and* set custom DTMCS/DMI IR values for RISC-V JTAG DTMs. This
|
||||
// means with the right config file, you can access a debug module hung from
|
||||
// the ECP5 TAP in this fashion using only upstream OpenOCD and gdb.
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_ecp5_jtag_dtm #(
|
||||
parameter DTMCS_IDLE_HINT = 3'd4,
|
||||
parameter W_PADDR = 9,
|
||||
parameter ABITS = W_PADDR - 2 // do not modify
|
||||
) (
|
||||
// This is synchronous to TCK and asserted for one TCK cycle only
|
||||
output wire dmihardreset_req,
|
||||
|
||||
// Bus clock + reset for Debug Module Interface
|
||||
input wire clk_dmi,
|
||||
input wire rst_n_dmi,
|
||||
|
||||
// Debug Module Interface (APB)
|
||||
output wire dmi_psel,
|
||||
output wire dmi_penable,
|
||||
output wire dmi_pwrite,
|
||||
output wire [W_PADDR-1:0] dmi_paddr,
|
||||
output wire [31:0] dmi_pwdata,
|
||||
input wire [31:0] dmi_prdata,
|
||||
input wire dmi_pready,
|
||||
input wire dmi_pslverr
|
||||
);
|
||||
|
||||
// Signals to/from the ECP5 TAP
|
||||
|
||||
wire jtdo2;
|
||||
wire jtdo1;
|
||||
wire jtdi;
|
||||
wire jtck_posedge_dont_use;
|
||||
wire jshift;
|
||||
wire jupdate;
|
||||
wire jrst_n;
|
||||
wire jce2;
|
||||
wire jce1;
|
||||
|
||||
JTAGG jtag_u (
|
||||
.JTDO2 (jtdo2),
|
||||
.JTDO1 (jtdo1),
|
||||
.JTDI (jtdi),
|
||||
.JTCK (jtck_posedge_dont_use),
|
||||
.JRTI2 (/* unused */),
|
||||
.JRTI1 (/* unused */),
|
||||
.JSHIFT (jshift),
|
||||
.JUPDATE (jupdate),
|
||||
.JRSTN (jrst_n),
|
||||
.JCE2 (jce2),
|
||||
.JCE1 (jce1)
|
||||
);
|
||||
|
||||
// JTAGG primitive asserts its signals synchronously to JTCK's posedge, but
|
||||
// you get weird and inconsistent results if you try to consume them
|
||||
// synchronously on JTCK's posedge, possibly due to a lack of hold
|
||||
// constraints in nextpnr.
|
||||
//
|
||||
// A quick hack is to move the sampling onto the negedge of the clock. This
|
||||
// then creates more problems because we would be running our shift logic on
|
||||
// a different edge from the control + CDC logic in the DTM core.
|
||||
//
|
||||
// So, even worse hack, move all our JTAG-domain logic onto the negedge
|
||||
// (or near enough) by inverting the clock.
|
||||
|
||||
wire jtck = !jtck_posedge_dont_use;
|
||||
|
||||
localparam W_DR_SHIFT = ABITS + 32 + 2;
|
||||
|
||||
reg core_dr_wen;
|
||||
reg core_dr_ren;
|
||||
reg core_dr_sel_dmi_ndtmcs;
|
||||
reg dr_shift_en;
|
||||
wire [W_DR_SHIFT-1:0] core_dr_wdata;
|
||||
wire [W_DR_SHIFT-1:0] core_dr_rdata;
|
||||
|
||||
// Decode our shift controls from the interesting ECP5 ones, and re-register
|
||||
// onto JTCK negedge (our posedge). Note without re-registering we observe
|
||||
// them a half-cycle (effectively one cycle) too early. This is another
|
||||
// consequence of the stupid JTDI thing
|
||||
|
||||
always @ (posedge jtck or negedge jrst_n) begin
|
||||
if (!jrst_n) begin
|
||||
core_dr_sel_dmi_ndtmcs <= 1'b0;
|
||||
core_dr_wen <= 1'b0;
|
||||
core_dr_ren <= 1'b0;
|
||||
dr_shift_en <= 1'b0;
|
||||
end else begin
|
||||
if (jce1 || jce2)
|
||||
core_dr_sel_dmi_ndtmcs <= jce2;
|
||||
core_dr_ren <= (jce1 || jce2) && !jshift;
|
||||
core_dr_wen <= jupdate;
|
||||
dr_shift_en <= jshift;
|
||||
end
|
||||
end
|
||||
|
||||
reg [W_DR_SHIFT-1:0] dr_shift;
|
||||
assign core_dr_wdata = dr_shift;
|
||||
|
||||
always @ (posedge jtck or negedge jrst_n) begin
|
||||
if (!jrst_n) begin
|
||||
dr_shift <= {W_DR_SHIFT{1'b0}};
|
||||
end else if (core_dr_ren) begin
|
||||
dr_shift <= core_dr_rdata;
|
||||
end else if (dr_shift_en) begin
|
||||
dr_shift <= {jtdi, dr_shift[W_DR_SHIFT-1:1]};
|
||||
if (!core_dr_sel_dmi_ndtmcs)
|
||||
dr_shift[31] <= jtdi;
|
||||
end
|
||||
end
|
||||
|
||||
// Not documented on ECP5: as well as the posedge flop on JTDI, the ECP5 puts
|
||||
// a negedge flop on JTDO1, JTDO2. (Conjecture based on dicking around with a
|
||||
// logic analyser.) To get JTDOx to appear with the same timing as our shifter
|
||||
// LSB (which we update on every JTCK negedge) we:
|
||||
//
|
||||
// - Register the LSB of the *next* value of dr_shift on the JTCK posedge, so
|
||||
// half a cycle earlier than the actual dr_shift update
|
||||
//
|
||||
// - This then gets re-registered with the pointless JTDO negedge flops, so
|
||||
// that it appears with the same timing as our DR shifter update.
|
||||
|
||||
reg dr_shift_next_halfcycle;
|
||||
always @ (negedge jtck or negedge jrst_n) begin
|
||||
if (!jrst_n) begin
|
||||
dr_shift_next_halfcycle <= 1'b0;
|
||||
end else begin
|
||||
dr_shift_next_halfcycle <=
|
||||
core_dr_ren ? core_dr_rdata[0] :
|
||||
dr_shift_en ? dr_shift[1] : dr_shift[0];
|
||||
end
|
||||
end
|
||||
|
||||
// We have only a single shifter for the ER1 and ER2 chains, so these are tied
|
||||
// together:
|
||||
|
||||
assign jtdo1 = dr_shift_next_halfcycle;
|
||||
assign jtdo2 = dr_shift_next_halfcycle;
|
||||
|
||||
// The actual DTM is in here:
|
||||
|
||||
hazard3_jtag_dtm_core #(
|
||||
.DTMCS_IDLE_HINT (DTMCS_IDLE_HINT),
|
||||
.W_ADDR (ABITS)
|
||||
) inst_hazard3_jtag_dtm_core (
|
||||
.tck (jtck),
|
||||
.trst_n (jrst_n),
|
||||
|
||||
.clk_dmi (clk_dmi),
|
||||
.rst_n_dmi (rst_n_dmi),
|
||||
|
||||
.dr_wen (core_dr_wen),
|
||||
.dr_ren (core_dr_ren),
|
||||
.dr_sel_dmi_ndtmcs (core_dr_sel_dmi_ndtmcs),
|
||||
.dr_wdata (core_dr_wdata),
|
||||
.dr_rdata (core_dr_rdata),
|
||||
|
||||
.dmihardreset_req (dmihardreset_req),
|
||||
|
||||
.dmi_psel (dmi_psel),
|
||||
.dmi_penable (dmi_penable),
|
||||
.dmi_pwrite (dmi_pwrite),
|
||||
.dmi_paddr (dmi_paddr[W_PADDR-1:2]),
|
||||
.dmi_pwdata (dmi_pwdata),
|
||||
.dmi_prdata (dmi_prdata),
|
||||
.dmi_pready (dmi_pready),
|
||||
.dmi_pslverr (dmi_pslverr)
|
||||
);
|
||||
|
||||
assign dmi_paddr[1:0] = 2'b00;
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,6 @@
|
||||
file hazard3_jtag_dtm.v
|
||||
file hazard3_jtag_dtm_core.v
|
||||
|
||||
file ../cdc/hazard3_apb_async_bridge.v
|
||||
file ../cdc/hazard3_reset_sync.v
|
||||
file ../cdc/hazard3_sync_1bit.v
|
||||
+210
@@ -0,0 +1,210 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Implementation of standard RISC-V JTAG-DTM with an APB Debug Module
|
||||
// Interface. The TAP itself is clocked directly by JTAG TCK; a clock
|
||||
// crossing is instantiated internally between the TCK domain and the DMI bus
|
||||
// clock domain.
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_jtag_dtm #(
|
||||
parameter IDCODE = 32'h0000_0001,
|
||||
parameter DTMCS_IDLE_HINT = 3'd4,
|
||||
parameter W_PADDR = 9,
|
||||
parameter ABITS = W_PADDR - 2 // do not modify
|
||||
) (
|
||||
// Standard JTAG signals -- the JTAG hardware is clocked directly by TCK.
|
||||
input wire tck,
|
||||
input wire trst_n,
|
||||
input wire tms,
|
||||
input wire tdi,
|
||||
output reg tdo,
|
||||
|
||||
// This is synchronous to TCK and asserted for one TCK cycle only
|
||||
output wire dmihardreset_req,
|
||||
|
||||
// Bus clock + reset for Debug Module Interface
|
||||
input wire clk_dmi,
|
||||
input wire rst_n_dmi,
|
||||
|
||||
// Debug Module Interface (APB)
|
||||
output wire dmi_psel,
|
||||
output wire dmi_penable,
|
||||
output wire dmi_pwrite,
|
||||
output wire [W_PADDR-1:0] dmi_paddr,
|
||||
output wire [31:0] dmi_pwdata,
|
||||
input wire [31:0] dmi_prdata,
|
||||
input wire dmi_pready,
|
||||
input wire dmi_pslverr
|
||||
);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// TAP state machine
|
||||
|
||||
reg [3:0] tap_state;
|
||||
localparam S_RESET = 4'd0;
|
||||
localparam S_RUN_IDLE = 4'd1;
|
||||
localparam S_SELECT_DR = 4'd2;
|
||||
localparam S_CAPTURE_DR = 4'd3;
|
||||
localparam S_SHIFT_DR = 4'd4;
|
||||
localparam S_EXIT1_DR = 4'd5;
|
||||
localparam S_PAUSE_DR = 4'd6;
|
||||
localparam S_EXIT2_DR = 4'd7;
|
||||
localparam S_UPDATE_DR = 4'd8;
|
||||
localparam S_SELECT_IR = 4'd9;
|
||||
localparam S_CAPTURE_IR = 4'd10;
|
||||
localparam S_SHIFT_IR = 4'd11;
|
||||
localparam S_EXIT1_IR = 4'd12;
|
||||
localparam S_PAUSE_IR = 4'd13;
|
||||
localparam S_EXIT2_IR = 4'd14;
|
||||
localparam S_UPDATE_IR = 4'd15;
|
||||
|
||||
always @ (posedge tck or negedge trst_n) begin
|
||||
if (!trst_n) begin
|
||||
tap_state <= S_RESET;
|
||||
end else case(tap_state)
|
||||
S_RESET : tap_state <= tms ? S_RESET : S_RUN_IDLE ;
|
||||
S_RUN_IDLE : tap_state <= tms ? S_SELECT_DR : S_RUN_IDLE ;
|
||||
|
||||
S_SELECT_DR : tap_state <= tms ? S_SELECT_IR : S_CAPTURE_DR;
|
||||
S_CAPTURE_DR : tap_state <= tms ? S_EXIT1_DR : S_SHIFT_DR ;
|
||||
S_SHIFT_DR : tap_state <= tms ? S_EXIT1_DR : S_SHIFT_DR ;
|
||||
S_EXIT1_DR : tap_state <= tms ? S_UPDATE_DR : S_PAUSE_DR ;
|
||||
S_PAUSE_DR : tap_state <= tms ? S_EXIT2_DR : S_PAUSE_DR ;
|
||||
S_EXIT2_DR : tap_state <= tms ? S_UPDATE_DR : S_SHIFT_DR ;
|
||||
S_UPDATE_DR : tap_state <= tms ? S_SELECT_DR : S_RUN_IDLE ;
|
||||
|
||||
S_SELECT_IR : tap_state <= tms ? S_RESET : S_CAPTURE_IR;
|
||||
S_CAPTURE_IR : tap_state <= tms ? S_EXIT1_IR : S_SHIFT_IR ;
|
||||
S_SHIFT_IR : tap_state <= tms ? S_EXIT1_IR : S_SHIFT_IR ;
|
||||
S_EXIT1_IR : tap_state <= tms ? S_UPDATE_IR : S_PAUSE_IR ;
|
||||
S_PAUSE_IR : tap_state <= tms ? S_EXIT2_IR : S_PAUSE_IR ;
|
||||
S_EXIT2_IR : tap_state <= tms ? S_UPDATE_IR : S_SHIFT_IR ;
|
||||
S_UPDATE_IR : tap_state <= tms ? S_SELECT_DR : S_RUN_IDLE ;
|
||||
endcase
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Instruction register
|
||||
|
||||
localparam W_IR = 5;
|
||||
// All other encodings behave as BYPASS:
|
||||
localparam IR_IDCODE = 5'h01;
|
||||
localparam IR_DTMCS = 5'h10;
|
||||
localparam IR_DMI = 5'h11;
|
||||
|
||||
reg [W_IR-1:0] ir_shift;
|
||||
reg [W_IR-1:0] ir;
|
||||
|
||||
always @ (posedge tck or negedge trst_n) begin
|
||||
if (!trst_n) begin
|
||||
ir_shift <= {W_IR{1'b0}};
|
||||
ir <= IR_IDCODE;
|
||||
end else if (tap_state == S_RESET) begin
|
||||
ir_shift <= {W_IR{1'b0}};
|
||||
ir <= IR_IDCODE;
|
||||
end else if (tap_state == S_CAPTURE_IR) begin
|
||||
ir_shift <= ir;
|
||||
end else if (tap_state == S_SHIFT_IR) begin
|
||||
ir_shift <= {tdi, ir_shift[W_IR-1:1]};
|
||||
end else if (tap_state == S_UPDATE_IR) begin
|
||||
ir <= ir_shift;
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Data registers
|
||||
|
||||
// Shift register is sized to largest DR, which is DMI:
|
||||
// {addr[7:0], data[31:0], op[1:0]}
|
||||
localparam W_DR_SHIFT = ABITS + 32 + 2;
|
||||
|
||||
reg [W_DR_SHIFT-1:0] dr_shift;
|
||||
|
||||
// Signals to/from the DTM core, which implements the DTMCS and DMI registers
|
||||
wire core_dr_wen;
|
||||
wire core_dr_ren;
|
||||
wire core_dr_sel_dmi_ndtmcs;
|
||||
wire [W_DR_SHIFT-1:0] core_dr_wdata;
|
||||
wire [W_DR_SHIFT-1:0] core_dr_rdata;
|
||||
|
||||
always @ (posedge tck or negedge trst_n) begin
|
||||
if (!trst_n) begin
|
||||
dr_shift <= {W_DR_SHIFT{1'b0}};
|
||||
end else if (tap_state == S_SHIFT_DR) begin
|
||||
dr_shift <= {tdi, dr_shift[W_DR_SHIFT-1:1]};
|
||||
// Shorten DR shift chain according to IR
|
||||
if (ir == IR_DMI)
|
||||
dr_shift[W_DR_SHIFT - 1] <= tdi;
|
||||
else if (ir == IR_IDCODE || ir == IR_DTMCS)
|
||||
dr_shift[31] <= tdi;
|
||||
else // BYPASS
|
||||
dr_shift[0] <= tdi;
|
||||
end else if (tap_state == S_CAPTURE_DR) begin
|
||||
if (ir == IR_DMI || ir == IR_DTMCS) begin
|
||||
dr_shift <= core_dr_rdata;
|
||||
end else if (ir == IR_IDCODE) begin
|
||||
dr_shift <= {{W_DR_SHIFT-32{1'b0}}, IDCODE};
|
||||
end else begin // BYPASS
|
||||
dr_shift <= {W_DR_SHIFT{1'b0}};
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// Must retime shift data onto negedge before presenting on TDO
|
||||
|
||||
always @ (negedge tck or negedge trst_n) begin
|
||||
if (!trst_n) begin
|
||||
tdo <= 1'b0;
|
||||
end else begin
|
||||
tdo <= tap_state == S_SHIFT_IR ? ir_shift[0] :
|
||||
tap_state == S_SHIFT_DR ? dr_shift[0] : 1'b0;
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Core logic and bus interface
|
||||
|
||||
assign core_dr_sel_dmi_ndtmcs = ir == IR_DMI;
|
||||
assign core_dr_wen = (ir == IR_DMI || ir == IR_DTMCS) && tap_state == S_UPDATE_DR;
|
||||
assign core_dr_ren = (ir == IR_DMI || ir == IR_DTMCS) && tap_state == S_CAPTURE_DR;
|
||||
|
||||
assign core_dr_wdata = dr_shift;
|
||||
|
||||
hazard3_jtag_dtm_core #(
|
||||
.DTMCS_IDLE_HINT (DTMCS_IDLE_HINT),
|
||||
.W_ADDR (ABITS)
|
||||
) dtm_core (
|
||||
.tck (tck),
|
||||
.trst_n (trst_n),
|
||||
.clk_dmi (clk_dmi),
|
||||
.rst_n_dmi (rst_n_dmi),
|
||||
|
||||
.dmihardreset_req (dmihardreset_req),
|
||||
|
||||
.dr_wen (core_dr_wen),
|
||||
.dr_ren (core_dr_ren),
|
||||
.dr_sel_dmi_ndtmcs (core_dr_sel_dmi_ndtmcs),
|
||||
.dr_wdata (core_dr_wdata),
|
||||
.dr_rdata (core_dr_rdata),
|
||||
|
||||
.dmi_psel (dmi_psel),
|
||||
.dmi_penable (dmi_penable),
|
||||
.dmi_pwrite (dmi_pwrite),
|
||||
.dmi_paddr (dmi_paddr[W_PADDR-1:2]),
|
||||
.dmi_pwdata (dmi_pwdata),
|
||||
.dmi_prdata (dmi_prdata),
|
||||
.dmi_pready (dmi_pready),
|
||||
.dmi_pslverr (dmi_pslverr)
|
||||
);
|
||||
|
||||
assign dmi_paddr[1:0] = 2'b00;
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,185 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// DTMCS + DMI control logic, bus interface and bus clock domain crossing for
|
||||
// a standard RISC-V APB JTAG-DTM. Essentially everything apart from the
|
||||
// actual TAP controller, IR and shift registers. Instantiated by
|
||||
// hazard3_jtag_dtm.v.
|
||||
//
|
||||
// This core logic can be reused and connected to some other serial transport
|
||||
// or, for example, the ECP5 JTAGG primitive (see hazard5_ecp5_jtag_dtm.v)
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_jtag_dtm_core #(
|
||||
parameter DTMCS_IDLE_HINT = 3'd4,
|
||||
parameter W_ADDR = 8,
|
||||
parameter W_DR_SHIFT = W_ADDR + 32 + 2 // do not modify
|
||||
) (
|
||||
input wire tck,
|
||||
input wire trst_n,
|
||||
|
||||
input wire clk_dmi,
|
||||
input wire rst_n_dmi,
|
||||
|
||||
// DR capture/update (read/write) signals
|
||||
input wire dr_wen,
|
||||
input wire dr_ren,
|
||||
input wire dr_sel_dmi_ndtmcs,
|
||||
input wire [W_DR_SHIFT-1:0] dr_wdata,
|
||||
output wire [W_DR_SHIFT-1:0] dr_rdata,
|
||||
|
||||
// This is synchronous to TCK and asserted for one TCK cycle only
|
||||
output reg dmihardreset_req,
|
||||
|
||||
// Debug Module Interface (APB)
|
||||
output wire dmi_psel,
|
||||
output wire dmi_penable,
|
||||
output wire dmi_pwrite,
|
||||
output wire [W_ADDR-1:0] dmi_paddr,
|
||||
output wire [31:0] dmi_pwdata,
|
||||
input wire [31:0] dmi_prdata,
|
||||
input wire dmi_pready,
|
||||
input wire dmi_pslverr
|
||||
);
|
||||
|
||||
wire write_dmi = dr_wen && dr_sel_dmi_ndtmcs;
|
||||
wire write_dtmcs = dr_wen && !dr_sel_dmi_ndtmcs;
|
||||
wire read_dmi = dr_ren && dr_sel_dmi_ndtmcs;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// DMI bus adapter
|
||||
|
||||
reg [1:0] dmi_cmderr;
|
||||
reg dmi_busy;
|
||||
|
||||
// DTM-domain bus, connected to a matching DM-domain bus via an APB crossing:
|
||||
wire dtm_psel;
|
||||
wire dtm_penable;
|
||||
wire dtm_pwrite;
|
||||
wire [W_ADDR-1:0] dtm_paddr;
|
||||
wire [31:0] dtm_pwdata;
|
||||
wire [31:0] dtm_prdata;
|
||||
wire dtm_pready;
|
||||
wire dtm_pslverr;
|
||||
|
||||
// We are relying on some particular features of our APB clock crossing here
|
||||
// to save some registers:
|
||||
//
|
||||
// - The transfer is launched immediately when psel is seen, no need to
|
||||
// actually assert an access phase (as the standard allows the CDC to
|
||||
// assume that access immediately follows setup) and no need to maintain
|
||||
// pwrite/paddr/pwdata valid after the setup phase
|
||||
//
|
||||
// - prdata/pslverr remain valid after the transfer completes, until the next
|
||||
// transfer completes
|
||||
//
|
||||
// These allow us to connect the upstream side of the CDC directly to our DR
|
||||
// shifter without any sample/hold registers in between.
|
||||
|
||||
// psel is only pulsed for one cycle, penable is not asserted.
|
||||
assign dtm_psel = write_dmi &&
|
||||
(dr_wdata[1:0] == 2'd1 || dr_wdata[1:0] == 2'd2) &&
|
||||
!(dmi_busy || dmi_cmderr != 2'd0) && dtm_pready;
|
||||
assign dtm_penable = 1'b0;
|
||||
|
||||
// paddr/pwdata/pwrite are valid momentarily when psel is asserted.
|
||||
assign dtm_paddr = dr_wdata[34 +: W_ADDR];
|
||||
assign dtm_pwrite = dr_wdata[1];
|
||||
assign dtm_pwdata = dr_wdata[2 +: 32];
|
||||
|
||||
always @ (posedge tck or negedge trst_n) begin
|
||||
if (!trst_n) begin
|
||||
dmi_busy <= 1'b0;
|
||||
dmi_cmderr <= 2'd0;
|
||||
end else if (read_dmi) begin
|
||||
// Reading while busy sets the busy sticky error. Note the capture
|
||||
// into shift register should also reflect this update on-the-fly
|
||||
if (dmi_busy && dmi_cmderr == 2'd0)
|
||||
dmi_cmderr <= 2'h3;
|
||||
end else if (write_dtmcs) begin
|
||||
// Writing dtmcs.dmireset = 1 clears a sticky error
|
||||
if (dr_wdata[16])
|
||||
dmi_cmderr <= 2'd0;
|
||||
end else if (write_dmi) begin
|
||||
if (dtm_psel) begin
|
||||
dmi_busy <= 1'b1;
|
||||
end else if (dr_wdata[1:0] != 2'd0) begin
|
||||
// DMI ignored operation, so set sticky busy
|
||||
if (dmi_cmderr == 2'd0)
|
||||
dmi_cmderr <= 2'd3;
|
||||
end
|
||||
end else if (dmi_busy && dtm_pready) begin
|
||||
dmi_busy <= 1'b0;
|
||||
if (dmi_cmderr == 2'd0 && dtm_pslverr)
|
||||
dmi_cmderr <= 2'd2;
|
||||
end
|
||||
end
|
||||
|
||||
// DTM logic is in TCK domain, actual DMI + DM is in processor domain
|
||||
|
||||
hazard3_apb_async_bridge #(
|
||||
.W_ADDR (W_ADDR),
|
||||
.W_DATA (32),
|
||||
.N_SYNC_STAGES (2)
|
||||
) inst_hazard3_apb_async_bridge (
|
||||
.clk_src (tck),
|
||||
.rst_n_src (trst_n),
|
||||
|
||||
.clk_dst (clk_dmi),
|
||||
.rst_n_dst (rst_n_dmi),
|
||||
|
||||
.src_psel (dtm_psel),
|
||||
.src_penable (dtm_penable),
|
||||
.src_pwrite (dtm_pwrite),
|
||||
.src_paddr (dtm_paddr),
|
||||
.src_pwdata (dtm_pwdata),
|
||||
.src_prdata (dtm_prdata),
|
||||
.src_pready (dtm_pready),
|
||||
.src_pslverr (dtm_pslverr),
|
||||
|
||||
.dst_psel (dmi_psel),
|
||||
.dst_penable (dmi_penable),
|
||||
.dst_pwrite (dmi_pwrite),
|
||||
.dst_paddr (dmi_paddr),
|
||||
.dst_pwdata (dmi_pwdata),
|
||||
.dst_prdata (dmi_prdata),
|
||||
.dst_pready (dmi_pready),
|
||||
.dst_pslverr (dmi_pslverr)
|
||||
);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// DR read/write
|
||||
|
||||
wire [W_DR_SHIFT-1:0] dtmcs_rdata = {
|
||||
{W_ADDR{1'b0}},
|
||||
19'h0,
|
||||
DTMCS_IDLE_HINT[2:0],
|
||||
dmi_cmderr,
|
||||
W_ADDR[5:0], // abits
|
||||
4'd1 // version
|
||||
};
|
||||
|
||||
wire [W_DR_SHIFT-1:0] dmi_rdata = {
|
||||
{W_ADDR{1'b0}},
|
||||
dtm_prdata,
|
||||
dmi_busy && dmi_cmderr == 2'd0 ? 2'd3 : dmi_cmderr
|
||||
};
|
||||
|
||||
assign dr_rdata = dr_sel_dmi_ndtmcs ? dmi_rdata : dtmcs_rdata;
|
||||
|
||||
always @ (posedge tck or negedge trst_n) begin
|
||||
if (!trst_n) begin
|
||||
dmihardreset_req <= 1'b0;
|
||||
end else begin
|
||||
dmihardreset_req <= write_dtmcs && dr_wdata[17];
|
||||
end
|
||||
end
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
@@ -0,0 +1,162 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2025 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Implement a RISC-V JTAG DTM tunnelled through a Xilinx BSCANE2 primitive.
|
||||
//
|
||||
// Xilinx allows up to four custom DRs to be added to the FPGA TAP controller.
|
||||
// A JTAG-DTM only needs two: DTMCS and DMI.
|
||||
//
|
||||
// With the correct config, OpenOCD can treat the FPGA TAP as a JTAG DTM and
|
||||
// access RISC-V debug directly. This allows you to debug internal RISC-V
|
||||
// cores with the same JTAG interface you use to load the FPGA.
|
||||
//
|
||||
// CHAIN_DTMCS and CHAIN_DMI select which JTAG IR values are used to access
|
||||
// these DRs. Values 1 through 4 correspond to Xilinx USER1 through USER4
|
||||
// instructions, which have IR values 0x02, 0x03, 0x22, 0x23.
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_xilinx7_jtag_dtm #(
|
||||
parameter SEL_DTMCS = 3,
|
||||
parameter SEL_DMI = 4,
|
||||
parameter DTMCS_IDLE_HINT = 3'd4,
|
||||
parameter W_PADDR = 9,
|
||||
parameter ABITS = W_PADDR - 2 // do not modify
|
||||
) (
|
||||
// This is synchronous to TCK and asserted for one TCK cycle only
|
||||
output wire dmihardreset_req,
|
||||
|
||||
// Bus clock + reset for Debug Module Interface
|
||||
input wire clk_dmi,
|
||||
input wire rst_n_dmi,
|
||||
|
||||
// Debug Module Interface (APB)
|
||||
output wire dmi_psel,
|
||||
output wire dmi_penable,
|
||||
output wire dmi_pwrite,
|
||||
output wire [W_PADDR-1:0] dmi_paddr,
|
||||
output wire [31:0] dmi_pwdata,
|
||||
input wire [31:0] dmi_prdata,
|
||||
input wire dmi_pready,
|
||||
input wire dmi_pslverr
|
||||
);
|
||||
|
||||
// Signals to/from the Xilinx TAP
|
||||
|
||||
wire jtck_unbuf;
|
||||
wire jtck;
|
||||
wire jtdo2;
|
||||
wire jtdo1;
|
||||
wire jtdi;
|
||||
wire jshift;
|
||||
wire jupdate;
|
||||
wire jcapture;
|
||||
wire jrst;
|
||||
wire jrst_n = !jrst;
|
||||
wire jce2;
|
||||
wire jce1;
|
||||
|
||||
BSCANE2 #(
|
||||
.JTAG_CHAIN (SEL_DTMCS) // Value for USER command.
|
||||
) bscan_dtmcs (
|
||||
.CAPTURE (jcapture), // CAPTURE output from TAP controller.
|
||||
.DRCK (/* unused */), // Gated TCK output. When SEL is asserted, DRCK toggles when CAPTURE or SHIFT are asserted.
|
||||
.RESET (jrst), // Reset output for TAP controller.
|
||||
.RUNTEST (/* unused */), // Output asserted when TAP controller is in Run Test/Idle state.
|
||||
.SEL (jce1), // USER instruction active output.
|
||||
.SHIFT (jshift), // SHIFT output from TAP controller.
|
||||
.TCK (jtck_unbuf), // Test Clock output. Fabric connection to TAP Clock pin.
|
||||
.TDI (jtdi), // Test Data Input (TDI) output from TAP controller.
|
||||
.TMS (/* unused */), // Test Mode Select output. Fabric connection to TAP.
|
||||
.UPDATE (jupdate), // UPDATE output from TAP controller
|
||||
.TDO (jtdo1) // Test Data Output (TDO) input for USER function.
|
||||
);
|
||||
|
||||
BSCANE2 #(
|
||||
.JTAG_CHAIN (SEL_DMI)
|
||||
) bscan_dmi (
|
||||
.CAPTURE (/* unused */),
|
||||
.DRCK (/* unused */),
|
||||
.RESET (/* unused */),
|
||||
.RUNTEST (/* unused */),
|
||||
.SEL (jce2),
|
||||
.SHIFT (/* unused */),
|
||||
.TCK (/* unused */),
|
||||
.TDI (/* unused */),
|
||||
.TMS (/* unused */),
|
||||
.UPDATE (/* unused */),
|
||||
.TDO (jtdo2)
|
||||
);
|
||||
|
||||
BUFG bufg_jtck (
|
||||
.I (jtck_unbuf),
|
||||
.O (jtck)
|
||||
);
|
||||
|
||||
localparam W_DR_SHIFT = ABITS + 32 + 2;
|
||||
|
||||
wire core_dr_wen = jupdate;
|
||||
wire core_dr_ren = jcapture;
|
||||
wire core_dr_sel_dmi_ndtmcs = !jce1;
|
||||
wire dr_shift_en = jshift;
|
||||
wire [W_DR_SHIFT-1:0] core_dr_wdata;
|
||||
wire [W_DR_SHIFT-1:0] core_dr_rdata;
|
||||
|
||||
reg [W_DR_SHIFT-1:0] dr_shift;
|
||||
assign core_dr_wdata = dr_shift;
|
||||
|
||||
always @ (posedge jtck or negedge jrst_n) begin
|
||||
if (!jrst_n) begin
|
||||
dr_shift <= {W_DR_SHIFT{1'b0}};
|
||||
end else if (core_dr_ren) begin
|
||||
dr_shift <= core_dr_rdata;
|
||||
end else if (dr_shift_en) begin
|
||||
dr_shift <= {jtdi, dr_shift[W_DR_SHIFT-1:1]};
|
||||
if (!core_dr_sel_dmi_ndtmcs)
|
||||
dr_shift[31] <= jtdi;
|
||||
end
|
||||
end
|
||||
|
||||
// We have only a single shifter for the two DRs, so these are tied together:
|
||||
assign jtdo1 = dr_shift[0];
|
||||
assign jtdo2 = dr_shift[0];
|
||||
|
||||
// The actual DTM is in here:
|
||||
|
||||
hazard3_jtag_dtm_core #(
|
||||
.DTMCS_IDLE_HINT (DTMCS_IDLE_HINT),
|
||||
.W_ADDR (ABITS)
|
||||
) inst_hazard3_jtag_dtm_core (
|
||||
.tck (jtck),
|
||||
.trst_n (jrst_n),
|
||||
|
||||
.clk_dmi (clk_dmi),
|
||||
.rst_n_dmi (rst_n_dmi),
|
||||
|
||||
.dr_wen (core_dr_wen),
|
||||
.dr_ren (core_dr_ren),
|
||||
.dr_sel_dmi_ndtmcs (core_dr_sel_dmi_ndtmcs),
|
||||
.dr_wdata (core_dr_wdata),
|
||||
.dr_rdata (core_dr_rdata),
|
||||
|
||||
.dmihardreset_req (dmihardreset_req),
|
||||
|
||||
.dmi_psel (dmi_psel),
|
||||
.dmi_penable (dmi_penable),
|
||||
.dmi_pwrite (dmi_pwrite),
|
||||
.dmi_paddr (dmi_paddr[W_PADDR-1:2]),
|
||||
.dmi_pwdata (dmi_pwdata),
|
||||
.dmi_prdata (dmi_prdata),
|
||||
.dmi_pready (dmi_pready),
|
||||
.dmi_pslverr (dmi_pslverr)
|
||||
);
|
||||
|
||||
assign dmi_paddr[1:0] = 2'b00;
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
Vendored
+22
@@ -0,0 +1,22 @@
|
||||
file hazard3_core.v
|
||||
file hazard3_cpu_1port.v
|
||||
file hazard3_cpu_2port.v
|
||||
file arith/hazard3_alu.v
|
||||
file arith/hazard3_branchcmp.v
|
||||
file arith/hazard3_mul_fast.v
|
||||
file arith/hazard3_muldiv_seq.v
|
||||
file arith/hazard3_onehot_encode.v
|
||||
file arith/hazard3_onehot_priority.v
|
||||
file arith/hazard3_onehot_priority_dynamic.v
|
||||
file arith/hazard3_priority_encode.v
|
||||
file arith/hazard3_shift_barrel.v
|
||||
file hazard3_csr.v
|
||||
file hazard3_decode.v
|
||||
file hazard3_frontend.v
|
||||
file hazard3_instr_decompress.v
|
||||
file hazard3_irq_ctrl.v
|
||||
file hazard3_pmp.v
|
||||
file hazard3_power_ctrl.v
|
||||
file hazard3_regfile_1w2r.v
|
||||
file hazard3_triggers.v
|
||||
include .
|
||||
Vendored
+259
@@ -0,0 +1,259 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Hazard3 CPU configuration parameters
|
||||
|
||||
// To configure Hazard3 you can either edit this file, or set parameters on
|
||||
// your top-level instantiation, it's up to you. These parameters are all
|
||||
// plumbed through Hazard3's internal hierarchy to the appropriate places.
|
||||
|
||||
// If you add a parameter here, you should add a matching line to
|
||||
// hazard3_config_inst.vh to propagate the parameter through module
|
||||
// instantiations.
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Reset state configuration
|
||||
|
||||
// RESET_VECTOR: Address of first instruction executed.
|
||||
parameter RESET_VECTOR = 32'h00000000,
|
||||
|
||||
// MTVEC_INIT: Initial value of trap vector base. Bits clear in MTVEC_WMASK
|
||||
// will never change from this initial value. Bits set in MTVEC_WMASK can be
|
||||
// written/set/cleared as normal.
|
||||
//
|
||||
// Note that mtvec bits 1:0 do not affect the trap base (as per RISC-V spec).
|
||||
// Bit 1 is don't care, bit 0 selects the vectoring mode: unvectored if == 0
|
||||
// (all traps go to mtvec), vectored if == 1 (exceptions go to mtvec, IRQs to
|
||||
// mtvec + mcause * 4). This means MTVEC_INIT also sets the initial vectoring
|
||||
// mode.
|
||||
parameter MTVEC_INIT = 32'h00000000,
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Standard RISC-V ISA support
|
||||
|
||||
// EXTENSION_A: Support for atomic read/modify/write instructions
|
||||
parameter EXTENSION_A = 1,
|
||||
|
||||
// EXTENSION_C: Support for compressed (variable-width) instructions
|
||||
parameter EXTENSION_C = 1,
|
||||
|
||||
// EXTENSION_E: Implement the RV32E base extension rather than RV32I. This
|
||||
// reduces the number of integer registers from 31 to 15.
|
||||
parameter EXTENSION_E = 0,
|
||||
|
||||
// EXTENSION_M: Support for hardware multiply/divide/modulo instructions
|
||||
parameter EXTENSION_M = 1,
|
||||
|
||||
// EXTENSION_ZBA: Support for Zba address generation instructions
|
||||
parameter EXTENSION_ZBA = 0,
|
||||
|
||||
// EXTENSION_ZBB: Support for Zbb basic bit manipulation instructions
|
||||
parameter EXTENSION_ZBB = 0,
|
||||
|
||||
// EXTENSION_ZBC: Support for Zbc carry-less multiplication instructions
|
||||
parameter EXTENSION_ZBC = 0,
|
||||
|
||||
// EXTENSION_ZBKB: Support for Zbkb basic bit manipulation for cryptography
|
||||
// Requires: Zbb. (This flag enables instructions in Zbkb which aren't in Zbb.)
|
||||
parameter EXTENSION_ZBKB = 0,
|
||||
|
||||
// EXTENSION_ZBKX: support for Zbkx crossbar permutation instructions
|
||||
parameter EXTENSION_ZBKX = 0,
|
||||
|
||||
// EXTENSION_ZBS: Support for Zbs single-bit manipulation instructions
|
||||
parameter EXTENSION_ZBS = 0,
|
||||
|
||||
// EXTENSION_ZCB: Support for Zcb basic additional compressed instructions
|
||||
// Requires: EXTENSION_C. (Some Zcb instructions also require Zbb or M.)
|
||||
// Note Zca is equivalent to C, as we do not support the F extension.
|
||||
parameter EXTENSION_ZCB = 0,
|
||||
|
||||
// EXTENSION_ZCLSD: Support for Zclsd compressed load/store pair instructions
|
||||
// Requires: EXTENSION_ZILSD, EXTENSION_C.
|
||||
parameter EXTENSION_ZCLSD = 0,
|
||||
|
||||
// EXTENSION_ZCMP: Support for Zcmp push/pop instructions.
|
||||
// Requires: EXTENSION_C.
|
||||
parameter EXTENSION_ZCMP = 0,
|
||||
|
||||
// EXTENSION_ZIFENCEI: Support for the fence.i instruction
|
||||
// Optional, since a plain branch/jump will also flush the prefetch queue.
|
||||
parameter EXTENSION_ZIFENCEI = 0,
|
||||
|
||||
// EXTENSION_ZILSD: Support for Zilsd load/store pair instructions
|
||||
parameter EXTENSION_ZILSD = 0,
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Custom RISC-V extensions
|
||||
|
||||
// EXTENSION_XH3B: Custom bit-extract-multiple instructions for Hazard3
|
||||
parameter EXTENSION_XH3BEXTM = 0,
|
||||
|
||||
// EXTENSION_XH3IRQ: Custom preemptive, prioritised interrupt support. Can be
|
||||
// disabled if an external interrupt controller (e.g. PLIC) is used. If
|
||||
// disabled, and NUM_IRQS > 1, the external interrupts are simply OR'd into
|
||||
// mip.meip.
|
||||
parameter EXTENSION_XH3IRQ = 0,
|
||||
|
||||
// EXTENSION_XH3PMPM: PMPCFGMx CSRs to enforce PMP regions in M-mode without
|
||||
// locking. Unlike ePMP mseccfg.rlb, locked and unlocked regions can coexist
|
||||
parameter EXTENSION_XH3PMPM = 0,
|
||||
|
||||
// EXTENSION_XH3POWER: Custom power management controls for Hazard3
|
||||
parameter EXTENSION_XH3POWER = 0,
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Standard CSR support
|
||||
|
||||
// Note the Zicsr extension is implied by any of CSR_M_MANDATORY, CSR_M_TRAP,
|
||||
// CSR_COUNTER.
|
||||
|
||||
// CSR_M_MANDATORY: Bare minimum CSR support e.g. misa. Spec says must = 1 if
|
||||
// CSRs are present, but I won't tell anyone.
|
||||
parameter CSR_M_MANDATORY = 1,
|
||||
|
||||
// CSR_M_TRAP: Include M-mode trap-handling CSRs, and enable trap support.
|
||||
parameter CSR_M_TRAP = 1,
|
||||
|
||||
// CSR_COUNTER: Include performance counters and Zicntr CSRs
|
||||
parameter CSR_COUNTER = 0,
|
||||
|
||||
// U_MODE: Support the U (user) execution mode. In U mode, the core performs
|
||||
// unprivileged bus accesses, and software's access to CSRs is restricted.
|
||||
// Additionally, if the PMP is included, the core may restrict U-mode
|
||||
// software's access to memory.
|
||||
// Requires: CSR_M_TRAP.
|
||||
parameter U_MODE = 0,
|
||||
|
||||
// PMP_REGIONS: Number of physical memory protection regions, or 0 for no PMP.
|
||||
// PMP is more useful if U mode is supported, but this is not a requirement.
|
||||
parameter PMP_REGIONS = 0,
|
||||
|
||||
// PMP_GRAIN: This is the "G" parameter in the privileged spec. Minimum PMP
|
||||
// region size is 1 << (G + 2) bytes. If G > 0, PMCFG.A can not be set to
|
||||
// NA4 (will get set to OFF instead). If G > 1, the G - 1 LSBs of pmpaddr are
|
||||
// read-only-0 when PMPCFG.A is OFF, and read-only-1 when PMPCFG.A is NAPOT.
|
||||
parameter PMP_GRAIN = 0,
|
||||
|
||||
// PMP_MATCH_NAPOT: Enable PMP support for the NAPOT (naturally-aligned
|
||||
// power-of-two) and NA4 (naturally-aligned four-byte) matching modes. When
|
||||
// disabled, attempting to select these modes will set the PMP region to OFF.
|
||||
parameter PMP_MATCH_NAPOT = 1,
|
||||
|
||||
// PMP_MATCH_TOR: Enable PMP support for the TOR (top-of-range) matching mode.
|
||||
// When disabled, attempting to select this mode will set the region to OFF.
|
||||
parameter PMP_MATCH_TOR = 0,
|
||||
|
||||
// PMPADDR_HARDWIRED: If a bit is 1, the corresponding region's pmpaddr and
|
||||
// pmpcfg registers are read-only. PMP_GRAIN is ignored on hardwired regions.
|
||||
// It's recommended to make hardwired regions the highest-numbered, so they
|
||||
// can be overridden by lower-numbered regions.
|
||||
parameter PMP_HARDWIRED = {(PMP_REGIONS > 0 ? PMP_REGIONS : 1){1'b0}},
|
||||
|
||||
// PMPADDR_HARDWIRED_ADDR: Values of pmpaddr registers whose PMP_HARDWIRED
|
||||
// bits are set to 1. Non-hardwired regions reset to all-zeroes.
|
||||
parameter PMP_HARDWIRED_ADDR = {(PMP_REGIONS > 0 ? PMP_REGIONS : 1){32'h0}},
|
||||
|
||||
// PMPCFG_RESET_VAL: Values of pmpcfg registers whose PMP_HARDWIRED bits are
|
||||
// set to 1. Non-hardwired regions reset to all zeroes.
|
||||
parameter PMP_HARDWIRED_CFG = {(PMP_REGIONS > 0 ? PMP_REGIONS : 1){8'h00}},
|
||||
|
||||
// DEBUG_SUPPORT: Support for run/halt and instruction injection from an
|
||||
// external Debug Module, support for Debug Mode, and Debug Mode CSRs.
|
||||
// Requires: CSR_M_MANDATORY, CSR_M_TRAP.
|
||||
parameter DEBUG_SUPPORT = 0,
|
||||
|
||||
// BREAKPOINT_TRIGGERS: Number of triggers which support type=2 execute=1
|
||||
// (but not store/load=1, i.e. not a watchpoint). Requires: DEBUG_SUPPORT
|
||||
parameter BREAKPOINT_TRIGGERS = 0,
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// External interrupt support
|
||||
|
||||
// NUM_IRQS: Number of external IRQs. Minimum 1, maximum 512. Note that if
|
||||
// EXTENSION_XH3IRQ (Hazard3 interrupt controller) is disabled then multiple
|
||||
// external interrupts are simply OR'd into mip.meip.
|
||||
parameter NUM_IRQS = 1,
|
||||
|
||||
// IRQ_PRIORITY_BITS: Number of priority bits implemented for each interrupt
|
||||
// in meipra, if EXTENSION_XH3IRQ is enabled. The number of distinct levels
|
||||
// is (1 << IRQ_PRIORITY_BITS). Minimum 0, max 4. Note that multiple priority
|
||||
// levels with a large number of IRQs will have a severe effect on timing.
|
||||
parameter IRQ_PRIORITY_BITS = 0,
|
||||
|
||||
// IRQ_INPUT_BYPASS: disable the input registers on the external interrupts,
|
||||
// to reduce latency by one cycle. Can be applied on an IRQ-by-IRQ basis.
|
||||
// Ignored if EXTENSION_XH3IRQ is disabled.
|
||||
parameter IRQ_INPUT_BYPASS = {(NUM_IRQS > 0 ? NUM_IRQS : 1){1'b0}},
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// ID registers
|
||||
|
||||
// JEDEC JEP106-compliant vendor ID, can be left at 0 if "not implemented or
|
||||
// [...] this is a non-commercial implementation" (RISC-V spec).
|
||||
// 31:7 is continuation code count, 6:0 is ID. Parity bit is not stored.
|
||||
parameter MVENDORID_VAL = 32'h0,
|
||||
|
||||
// Pointer to configuration structure blob, or all-zeroes. Must be at least
|
||||
// 4-byte-aligned.
|
||||
parameter MCONFIGPTR_VAL = 32'h0,
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Performance/size options
|
||||
|
||||
// REDUCED_BYPASS: Remove all forwarding paths except X->X (so back-to-back
|
||||
// ALU ops can still run at 1 CPI), to save area.
|
||||
parameter REDUCED_BYPASS = 0,
|
||||
|
||||
// MULDIV_UNROLL: Bits per clock for multiply/divide circuit, if present. Must
|
||||
// be a power of 2.
|
||||
parameter MULDIV_UNROLL = 1,
|
||||
|
||||
// MUL_FAST: Use single-cycle multiply circuit for MUL instructions, retiring
|
||||
// to stage 3. The sequential multiply/divide circuit is still used for MULH*
|
||||
parameter MUL_FAST = 0,
|
||||
|
||||
// MUL_FASTER: Retire fast multiply results to stage 2 instead of stage 3.
|
||||
// Throughput is the same, but latency is reduced from 2 cycles to 1 cycle.
|
||||
// Requires: MUL_FAST.
|
||||
parameter MUL_FASTER = 0,
|
||||
|
||||
// MULH_FAST: extend the fast multiply circuit to also cover MULH*, and remove
|
||||
// the multiply functionality from the sequential multiply/divide circuit.
|
||||
// Requires: MUL_FAST
|
||||
parameter MULH_FAST = 0,
|
||||
|
||||
// FAST_BRANCHCMP: Instantiate a separate comparator (eq/lt/ltu) for branch
|
||||
// comparisons, rather than using the ALU. Improves fetch address delay,
|
||||
// especially if Zba extension is enabled. Disabling may save area.
|
||||
parameter FAST_BRANCHCMP = 1,
|
||||
|
||||
// RESET_REGFILE: whether to support reset of the general purpose registers.
|
||||
// There are around 1k bits in the register file, so the reset can be
|
||||
// disabled e.g. to permit block-RAM inference on FPGA.
|
||||
parameter RESET_REGFILE = 0,
|
||||
|
||||
// BRANCH_PREDICTOR: enable branch prediction. The branch predictor consists
|
||||
// of a single BTB entry which is allocated on a taken backward branch, and
|
||||
// cleared on a mispredicted nontaken branch, a fence.i or a trap. Successful
|
||||
// prediction eliminates the 1-cyle fetch bubble on a taken branch, usually
|
||||
// making tight loops faster.
|
||||
parameter BRANCH_PREDICTOR = 0,
|
||||
|
||||
// MTVEC_WMASK: Mask of which bits in mtvec are writable. Full writability is
|
||||
// recommended, because a common idiom in setup code is to set mtvec just
|
||||
// past code that may trap, as a hardware "try {...} catch" block.
|
||||
//
|
||||
// - The vectoring mode can be made fixed by clearing the LSB of MTVEC_WMASK
|
||||
//
|
||||
// - In vectored mode, the vector table must be aligned to its size, rounded
|
||||
// up to a power of two.
|
||||
parameter MTVEC_WMASK = 32'hfffffffd,
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Port size parameters (do not modify)
|
||||
|
||||
parameter W_ADDR = 32, // Do not modify
|
||||
parameter W_DATA = 32 // Do not modify
|
||||
+59
@@ -0,0 +1,59 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Pass-through of parameters defined in hazard3_config.vh, so that these can
|
||||
// be set at instantiation rather than editing the config file, and will flow
|
||||
// correctly down through the hierarchy.
|
||||
|
||||
.RESET_VECTOR (RESET_VECTOR),
|
||||
.MTVEC_INIT (MTVEC_INIT),
|
||||
.EXTENSION_A (EXTENSION_A),
|
||||
.EXTENSION_C (EXTENSION_C),
|
||||
.EXTENSION_E (EXTENSION_E),
|
||||
.EXTENSION_M (EXTENSION_M),
|
||||
.EXTENSION_ZBA (EXTENSION_ZBA),
|
||||
.EXTENSION_ZBB (EXTENSION_ZBB),
|
||||
.EXTENSION_ZBC (EXTENSION_ZBC),
|
||||
.EXTENSION_ZBKB (EXTENSION_ZBKB),
|
||||
.EXTENSION_ZBKX (EXTENSION_ZBKX),
|
||||
.EXTENSION_ZBS (EXTENSION_ZBS),
|
||||
.EXTENSION_ZCB (EXTENSION_ZCB),
|
||||
.EXTENSION_ZCLSD (EXTENSION_ZCLSD),
|
||||
.EXTENSION_ZCMP (EXTENSION_ZCMP),
|
||||
.EXTENSION_ZIFENCEI (EXTENSION_ZIFENCEI),
|
||||
.EXTENSION_ZILSD (EXTENSION_ZILSD),
|
||||
.EXTENSION_XH3BEXTM (EXTENSION_XH3BEXTM),
|
||||
.EXTENSION_XH3IRQ (EXTENSION_XH3IRQ),
|
||||
.EXTENSION_XH3PMPM (EXTENSION_XH3PMPM),
|
||||
.EXTENSION_XH3POWER (EXTENSION_XH3POWER),
|
||||
.CSR_M_MANDATORY (CSR_M_MANDATORY),
|
||||
.CSR_M_TRAP (CSR_M_TRAP),
|
||||
.CSR_COUNTER (CSR_COUNTER),
|
||||
.U_MODE (U_MODE),
|
||||
.PMP_REGIONS (PMP_REGIONS),
|
||||
.PMP_GRAIN (PMP_GRAIN),
|
||||
.PMP_MATCH_NAPOT (PMP_MATCH_NAPOT),
|
||||
.PMP_MATCH_TOR (PMP_MATCH_TOR),
|
||||
.PMP_HARDWIRED (PMP_HARDWIRED),
|
||||
.PMP_HARDWIRED_ADDR (PMP_HARDWIRED_ADDR),
|
||||
.PMP_HARDWIRED_CFG (PMP_HARDWIRED_CFG),
|
||||
.DEBUG_SUPPORT (DEBUG_SUPPORT),
|
||||
.BREAKPOINT_TRIGGERS (BREAKPOINT_TRIGGERS),
|
||||
.NUM_IRQS (NUM_IRQS),
|
||||
.IRQ_PRIORITY_BITS (IRQ_PRIORITY_BITS),
|
||||
.IRQ_INPUT_BYPASS (IRQ_INPUT_BYPASS),
|
||||
.MVENDORID_VAL (MVENDORID_VAL),
|
||||
.MCONFIGPTR_VAL (MCONFIGPTR_VAL),
|
||||
.REDUCED_BYPASS (REDUCED_BYPASS),
|
||||
.MULDIV_UNROLL (MULDIV_UNROLL),
|
||||
.MUL_FAST (MUL_FAST),
|
||||
.MUL_FASTER (MUL_FASTER),
|
||||
.MULH_FAST (MULH_FAST),
|
||||
.FAST_BRANCHCMP (FAST_BRANCHCMP),
|
||||
.BRANCH_PREDICTOR (BRANCH_PREDICTOR),
|
||||
.MTVEC_WMASK (MTVEC_WMASK),
|
||||
.RESET_REGFILE (RESET_REGFILE),
|
||||
.W_ADDR (W_ADDR),
|
||||
.W_DATA (W_DATA)
|
||||
Vendored
+1590
File diff suppressed because it is too large
Load Diff
+343
@@ -0,0 +1,343 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Single-ported top level file for Hazard3 CPU. This file instantiates the
|
||||
// Hazard3 core, and arbitrates its instruction fetch and load/store signals
|
||||
// down to a single AHB5 master port.
|
||||
|
||||
`ifdef HAZARD3_RVFI_STANDALONE
|
||||
`include "hazard3_rvfi_standalone_defs.vh"
|
||||
`endif
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_cpu_1port #(
|
||||
`include "hazard3_config.vh"
|
||||
) (
|
||||
// Global signals
|
||||
input wire clk,
|
||||
input wire clk_always_on,
|
||||
input wire rst_n,
|
||||
|
||||
`ifdef RISCV_FORMAL
|
||||
`RVFI_OUTPUTS ,
|
||||
`endif
|
||||
|
||||
// Power control signals
|
||||
output wire pwrup_req,
|
||||
input wire pwrup_ack,
|
||||
output wire clk_en,
|
||||
output wire unblock_out,
|
||||
input wire unblock_in,
|
||||
|
||||
// AHB5 Master port
|
||||
output reg [W_ADDR-1:0] haddr,
|
||||
output reg hwrite,
|
||||
output reg [1:0] htrans,
|
||||
output reg [2:0] hsize,
|
||||
output wire [2:0] hburst,
|
||||
output reg [3:0] hprot,
|
||||
output wire hmastlock,
|
||||
output reg [7:0] hmaster,
|
||||
output reg hexcl,
|
||||
input wire hready,
|
||||
input wire hresp,
|
||||
input wire hexokay,
|
||||
output wire [W_DATA-1:0] hwdata,
|
||||
input wire [W_DATA-1:0] hrdata,
|
||||
|
||||
// Memory ordering signals
|
||||
output wire fence_i_vld,
|
||||
output wire fence_d_vld,
|
||||
input wire fence_rdy,
|
||||
|
||||
// Debugger run/halt control
|
||||
input wire dbg_req_halt,
|
||||
input wire dbg_req_halt_on_reset,
|
||||
input wire dbg_req_resume,
|
||||
output wire dbg_halted,
|
||||
output wire dbg_running,
|
||||
// Debugger access to data0 CSR
|
||||
input wire [W_DATA-1:0] dbg_data0_rdata,
|
||||
output wire [W_DATA-1:0] dbg_data0_wdata,
|
||||
output wire dbg_data0_wen,
|
||||
// Debugger instruction injection
|
||||
input wire [W_DATA-1:0] dbg_instr_data,
|
||||
input wire dbg_instr_data_vld,
|
||||
output wire dbg_instr_data_rdy,
|
||||
output wire dbg_instr_caught_exception,
|
||||
output wire dbg_instr_caught_ebreak,
|
||||
|
||||
// Optional debug system bus access patch-through
|
||||
input wire [W_ADDR-1:0] dbg_sbus_addr,
|
||||
input wire dbg_sbus_write,
|
||||
input wire [1:0] dbg_sbus_size,
|
||||
input wire dbg_sbus_vld,
|
||||
output wire dbg_sbus_rdy,
|
||||
output wire dbg_sbus_err,
|
||||
input wire [W_DATA-1:0] dbg_sbus_wdata,
|
||||
output wire [W_DATA-1:0] dbg_sbus_rdata,
|
||||
|
||||
// Identification CSR values
|
||||
input wire [W_DATA-1:0] mhartid_val,
|
||||
input wire [3:0] eco_version,
|
||||
|
||||
// Level-sensitive interrupt sources
|
||||
input wire [NUM_IRQS-1:0] irq, // -> mip.meip
|
||||
input wire soft_irq, // -> mip.msip
|
||||
input wire timer_irq // -> mip.mtip
|
||||
);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Processor core
|
||||
|
||||
// Instruction fetch signals
|
||||
wire core_aph_req_i;
|
||||
wire core_aph_panic_i;
|
||||
wire core_aph_ready_i;
|
||||
wire core_dph_ready_i;
|
||||
wire core_dph_err_i;
|
||||
|
||||
wire [W_ADDR-1:0] core_haddr_i;
|
||||
wire [2:0] core_hsize_i;
|
||||
wire core_priv_i;
|
||||
wire [W_DATA-1:0] core_rdata_i;
|
||||
|
||||
|
||||
// Load/store signals
|
||||
wire core_aph_req_d;
|
||||
wire core_aph_excl_d;
|
||||
wire core_aph_ready_d;
|
||||
wire core_dph_ready_d;
|
||||
wire core_dph_err_d;
|
||||
wire core_dph_exokay_d;
|
||||
|
||||
wire [W_ADDR-1:0] core_haddr_d;
|
||||
wire [2:0] core_hsize_d;
|
||||
wire core_priv_d;
|
||||
wire core_hwrite_d;
|
||||
wire [W_DATA-1:0] core_wdata_d;
|
||||
wire [W_DATA-1:0] core_rdata_d;
|
||||
|
||||
|
||||
hazard3_core #(
|
||||
`include "hazard3_config_inst.vh"
|
||||
) core (
|
||||
.clk (clk),
|
||||
.clk_always_on (clk_always_on),
|
||||
.rst_n (rst_n),
|
||||
|
||||
.pwrup_req (pwrup_req),
|
||||
.pwrup_ack (pwrup_ack),
|
||||
.clk_en (clk_en),
|
||||
.unblock_out (unblock_out),
|
||||
.unblock_in (unblock_in),
|
||||
|
||||
`ifdef RISCV_FORMAL
|
||||
`RVFI_CONN ,
|
||||
`endif
|
||||
|
||||
.bus_aph_req_i (core_aph_req_i),
|
||||
.bus_aph_panic_i (core_aph_panic_i),
|
||||
.bus_aph_ready_i (core_aph_ready_i),
|
||||
.bus_dph_ready_i (core_dph_ready_i),
|
||||
.bus_dph_err_i (core_dph_err_i),
|
||||
.bus_haddr_i (core_haddr_i),
|
||||
.bus_hsize_i (core_hsize_i),
|
||||
.bus_priv_i (core_priv_i),
|
||||
.bus_rdata_i (core_rdata_i),
|
||||
|
||||
.bus_aph_req_d (core_aph_req_d),
|
||||
.bus_aph_excl_d (core_aph_excl_d),
|
||||
.bus_aph_ready_d (core_aph_ready_d),
|
||||
.bus_dph_ready_d (core_dph_ready_d),
|
||||
.bus_dph_err_d (core_dph_err_d),
|
||||
.bus_dph_exokay_d (core_dph_exokay_d),
|
||||
.bus_haddr_d (core_haddr_d),
|
||||
.bus_hsize_d (core_hsize_d),
|
||||
.bus_priv_d (core_priv_d),
|
||||
.bus_hwrite_d (core_hwrite_d),
|
||||
.bus_wdata_d (core_wdata_d),
|
||||
.bus_rdata_d (core_rdata_d),
|
||||
|
||||
.fence_i_vld (fence_i_vld),
|
||||
.fence_d_vld (fence_d_vld),
|
||||
.fence_rdy (fence_rdy),
|
||||
|
||||
.dbg_req_halt (dbg_req_halt),
|
||||
.dbg_req_halt_on_reset (dbg_req_halt_on_reset),
|
||||
.dbg_req_resume (dbg_req_resume),
|
||||
.dbg_halted (dbg_halted),
|
||||
.dbg_running (dbg_running),
|
||||
.dbg_data0_rdata (dbg_data0_rdata),
|
||||
.dbg_data0_wdata (dbg_data0_wdata),
|
||||
.dbg_data0_wen (dbg_data0_wen),
|
||||
.dbg_instr_data (dbg_instr_data),
|
||||
.dbg_instr_data_vld (dbg_instr_data_vld),
|
||||
.dbg_instr_data_rdy (dbg_instr_data_rdy),
|
||||
.dbg_instr_caught_exception (dbg_instr_caught_exception),
|
||||
.dbg_instr_caught_ebreak (dbg_instr_caught_ebreak),
|
||||
|
||||
.mhartid_val (mhartid_val),
|
||||
.eco_version (eco_version),
|
||||
|
||||
.irq (irq),
|
||||
.soft_irq (soft_irq),
|
||||
.timer_irq (timer_irq)
|
||||
);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Arbitration state machine
|
||||
|
||||
wire bus_gnt_i;
|
||||
wire bus_gnt_d;
|
||||
wire bus_gnt_s;
|
||||
|
||||
reg bus_hold_aph;
|
||||
reg [2:0] bus_gnt_ids_prev;
|
||||
|
||||
// Note use of clk_always_on: SBA may use this arbiter to access the bus
|
||||
// whilst the core is asleep.
|
||||
|
||||
always @ (posedge clk_always_on or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
bus_hold_aph <= 1'b0;
|
||||
bus_gnt_ids_prev <= 3'h0;
|
||||
end else begin
|
||||
bus_hold_aph <= htrans[1] && !hready && !hresp;
|
||||
bus_gnt_ids_prev <= {bus_gnt_i, bus_gnt_d, bus_gnt_s};
|
||||
end
|
||||
end
|
||||
|
||||
// Debug SBA access is lower priority than load/store, but higher than
|
||||
// instruction fetch. This isn't ideal, but in a tight loop the core may be
|
||||
// performing an instruction fetch or load/store on every single cycle, and
|
||||
// this is a simple way to guarantee eventual success of debugger accesses. A
|
||||
// more complex way would be to add a "panic timer" to boost a stalled sbus
|
||||
// access over an instruction fetch.
|
||||
|
||||
// Note that, often, the sbus will be disconnected: it doesn't provide any
|
||||
// increase in debugger bus throughput compared with the program buffer and
|
||||
// autoexec. It's useful for "minimally intrusive" debug bus access(i.e. less
|
||||
// intrusive than halting the core and resuming it) e.g. for Segger RTT.
|
||||
|
||||
reg bus_active_dph_s;
|
||||
|
||||
assign {bus_gnt_i, bus_gnt_d, bus_gnt_s} =
|
||||
bus_hold_aph ? bus_gnt_ids_prev :
|
||||
core_aph_panic_i ? 3'b100 :
|
||||
core_aph_req_d ? 3'b010 :
|
||||
dbg_sbus_vld && !bus_active_dph_s ? 3'b001 :
|
||||
core_aph_req_i ? 3'b100 :
|
||||
3'b000 ;
|
||||
reg bus_active_dph_i;
|
||||
reg bus_active_dph_d;
|
||||
|
||||
always @ (posedge clk_always_on or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
bus_active_dph_i <= 1'b0;
|
||||
bus_active_dph_d <= 1'b0;
|
||||
bus_active_dph_s <= 1'b0;
|
||||
end else if (hready) begin
|
||||
bus_active_dph_i <= bus_gnt_i;
|
||||
bus_active_dph_d <= bus_gnt_d;
|
||||
bus_active_dph_s <= bus_gnt_s;
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Address phase request muxing
|
||||
|
||||
localparam HTRANS_IDLE = 2'b00;
|
||||
localparam HTRANS_NSEQ = 2'b10;
|
||||
|
||||
wire [3:0] hprot_data = {
|
||||
2'b00, // Noncacheable/nonbufferable
|
||||
core_priv_d, // Privileged or Normal as per core state
|
||||
1'b1 // Data access
|
||||
};
|
||||
|
||||
wire [3:0] hprot_instr = {
|
||||
2'b00, // Noncacheable/nonbufferable
|
||||
core_priv_i, // Privileged or Normal as per core state
|
||||
1'b0 // Instruction access
|
||||
};
|
||||
|
||||
wire [3:0] hprot_sbus = {
|
||||
2'b00, // Noncacheable/nonbufferable
|
||||
1'b1, // Always privileged
|
||||
1'b1 // Data access
|
||||
};
|
||||
|
||||
assign hburst = 3'b000; // HBURST_SINGLE
|
||||
assign hmastlock = 1'b0;
|
||||
|
||||
always @ (*) begin
|
||||
if (bus_gnt_s) begin
|
||||
htrans = HTRANS_NSEQ;
|
||||
hexcl = 1'b0;
|
||||
haddr = dbg_sbus_addr;
|
||||
hsize = {1'b0, dbg_sbus_size};
|
||||
hwrite = dbg_sbus_write;
|
||||
hprot = hprot_sbus;
|
||||
hmaster = 8'h01;
|
||||
end else if (bus_gnt_d) begin
|
||||
htrans = HTRANS_NSEQ;
|
||||
hexcl = core_aph_excl_d;
|
||||
haddr = core_haddr_d;
|
||||
hsize = core_hsize_d;
|
||||
hwrite = core_hwrite_d;
|
||||
hprot = hprot_data;
|
||||
hmaster = 8'h00;
|
||||
end else if (bus_gnt_i) begin
|
||||
htrans = HTRANS_NSEQ;
|
||||
hexcl = 1'b0;
|
||||
haddr = core_haddr_i;
|
||||
hsize = core_hsize_i;
|
||||
hwrite = 1'b0;
|
||||
hprot = hprot_instr;
|
||||
hmaster = 8'h00;
|
||||
end else begin
|
||||
htrans = HTRANS_IDLE;
|
||||
hexcl = 1'b0;
|
||||
haddr = {W_ADDR{1'b0}};
|
||||
hsize = 3'h0;
|
||||
hwrite = 1'b0;
|
||||
hprot = 4'h0;
|
||||
hmaster = 8'h00;
|
||||
end
|
||||
end
|
||||
|
||||
assign hwdata = bus_active_dph_s ? dbg_sbus_wdata : core_wdata_d;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Response routing
|
||||
|
||||
// Data buses directly connected
|
||||
assign core_rdata_d = hrdata;
|
||||
assign core_rdata_i = hrdata;
|
||||
assign dbg_sbus_rdata = hrdata;
|
||||
|
||||
// Handhshake based on grant and bus stall
|
||||
assign core_aph_ready_i = hready && bus_gnt_i;
|
||||
assign core_dph_ready_i = bus_active_dph_i && hready;
|
||||
assign core_dph_err_i = bus_active_dph_i && hresp;
|
||||
|
||||
// D-side errors are reported even when not ready, so that the core can make
|
||||
// use of the two-phase error response to cleanly squash a second load/store
|
||||
// chasing the faulting one down the pipeline.
|
||||
assign core_aph_ready_d = hready && bus_gnt_d;
|
||||
assign core_dph_ready_d = bus_active_dph_d && hready;
|
||||
assign core_dph_err_d = bus_active_dph_d && hresp;
|
||||
assign core_dph_exokay_d = bus_active_dph_d && hexokay;
|
||||
|
||||
assign dbg_sbus_err = bus_active_dph_s && hresp;
|
||||
assign dbg_sbus_rdy = bus_active_dph_s && hready;
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+327
@@ -0,0 +1,327 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Dual-ported top level file for Hazard3 CPU. This file instantiates the
|
||||
// Hazard3 core, and interfaces its instruction fetch and load/store signals
|
||||
// to a pair of AHB5 master ports.
|
||||
|
||||
`ifdef HAZARD3_RVFI_STANDALONE
|
||||
`include "hazard3_rvfi_standalone_defs.vh"
|
||||
`endif
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_cpu_2port #(
|
||||
`include "hazard3_config.vh"
|
||||
) (
|
||||
// Global signals
|
||||
input wire clk,
|
||||
input wire clk_always_on,
|
||||
input wire rst_n,
|
||||
|
||||
// Power control signals
|
||||
output wire pwrup_req,
|
||||
input wire pwrup_ack,
|
||||
output wire clk_en,
|
||||
output wire unblock_out,
|
||||
input wire unblock_in,
|
||||
|
||||
`ifdef RISCV_FORMAL
|
||||
`RVFI_OUTPUTS ,
|
||||
`endif
|
||||
|
||||
// Instruction fetch port
|
||||
output wire [W_ADDR-1:0] i_haddr,
|
||||
output wire i_hwrite,
|
||||
output wire [1:0] i_htrans,
|
||||
output wire [2:0] i_hsize,
|
||||
output wire [2:0] i_hburst,
|
||||
output wire [3:0] i_hprot,
|
||||
output wire i_hmastlock,
|
||||
output wire [7:0] i_hmaster,
|
||||
input wire i_hready,
|
||||
input wire i_hresp,
|
||||
output wire [W_DATA-1:0] i_hwdata,
|
||||
input wire [W_DATA-1:0] i_hrdata,
|
||||
|
||||
// Load/store port
|
||||
output wire [W_ADDR-1:0] d_haddr,
|
||||
output wire d_hwrite,
|
||||
output wire [1:0] d_htrans,
|
||||
output wire [2:0] d_hsize,
|
||||
output wire [2:0] d_hburst,
|
||||
output wire [3:0] d_hprot,
|
||||
output wire d_hmastlock,
|
||||
output wire [7:0] d_hmaster,
|
||||
output wire d_hexcl,
|
||||
input wire d_hready,
|
||||
input wire d_hresp,
|
||||
input wire d_hexokay,
|
||||
output wire [W_DATA-1:0] d_hwdata,
|
||||
input wire [W_DATA-1:0] d_hrdata,
|
||||
|
||||
// Memory ordering signals
|
||||
output wire fence_i_vld,
|
||||
output wire fence_d_vld,
|
||||
input wire fence_rdy,
|
||||
|
||||
// Debugger run/halt control
|
||||
input wire dbg_req_halt,
|
||||
input wire dbg_req_halt_on_reset,
|
||||
input wire dbg_req_resume,
|
||||
output wire dbg_halted,
|
||||
output wire dbg_running,
|
||||
// Debugger access to data0 CSR
|
||||
input wire [W_DATA-1:0] dbg_data0_rdata,
|
||||
output wire [W_DATA-1:0] dbg_data0_wdata,
|
||||
output wire dbg_data0_wen,
|
||||
// Debugger instruction injection
|
||||
input wire [W_DATA-1:0] dbg_instr_data,
|
||||
input wire dbg_instr_data_vld,
|
||||
output wire dbg_instr_data_rdy,
|
||||
output wire dbg_instr_caught_exception,
|
||||
output wire dbg_instr_caught_ebreak,
|
||||
// Optional debug system bus access patch-through
|
||||
input wire [W_ADDR-1:0] dbg_sbus_addr,
|
||||
input wire dbg_sbus_write,
|
||||
input wire [1:0] dbg_sbus_size,
|
||||
input wire dbg_sbus_vld,
|
||||
output wire dbg_sbus_rdy,
|
||||
output wire dbg_sbus_err,
|
||||
input wire [W_DATA-1:0] dbg_sbus_wdata,
|
||||
output wire [W_DATA-1:0] dbg_sbus_rdata,
|
||||
|
||||
// Identification CSR values
|
||||
input wire [W_DATA-1:0] mhartid_val,
|
||||
input wire [3:0] eco_version,
|
||||
|
||||
// Level-sensitive interrupt sources
|
||||
input wire [NUM_IRQS-1:0] irq, // -> mip.meip
|
||||
input wire soft_irq, // -> mip.msip
|
||||
input wire timer_irq // -> mip.mtip
|
||||
);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Processor core
|
||||
|
||||
// Instruction fetch signals
|
||||
wire core_aph_req_i;
|
||||
wire core_aph_ready_i;
|
||||
wire core_dph_ready_i;
|
||||
wire core_dph_err_i;
|
||||
|
||||
wire [W_ADDR-1:0] core_haddr_i;
|
||||
wire [2:0] core_hsize_i;
|
||||
wire core_priv_i;
|
||||
wire [W_DATA-1:0] core_rdata_i;
|
||||
|
||||
|
||||
// Load/store signals
|
||||
wire core_aph_req_d;
|
||||
wire core_aph_excl_d;
|
||||
wire core_aph_ready_d;
|
||||
wire core_dph_ready_d;
|
||||
wire core_dph_err_d;
|
||||
wire core_dph_exokay_d;
|
||||
|
||||
wire [W_ADDR-1:0] core_haddr_d;
|
||||
wire [2:0] core_hsize_d;
|
||||
wire core_priv_d;
|
||||
wire core_hwrite_d;
|
||||
wire [W_DATA-1:0] core_wdata_d;
|
||||
wire [W_DATA-1:0] core_rdata_d;
|
||||
|
||||
hazard3_core #(
|
||||
`include "hazard3_config_inst.vh"
|
||||
) core (
|
||||
.clk (clk),
|
||||
.clk_always_on (clk_always_on),
|
||||
.rst_n (rst_n),
|
||||
|
||||
.pwrup_req (pwrup_req),
|
||||
.pwrup_ack (pwrup_ack),
|
||||
.clk_en (clk_en),
|
||||
.unblock_out (unblock_out),
|
||||
.unblock_in (unblock_in),
|
||||
|
||||
`ifdef RISCV_FORMAL
|
||||
`RVFI_CONN ,
|
||||
`endif
|
||||
|
||||
.bus_aph_req_i (core_aph_req_i),
|
||||
.bus_aph_panic_i (/* unused for 2port */),
|
||||
.bus_aph_ready_i (core_aph_ready_i),
|
||||
.bus_dph_ready_i (core_dph_ready_i),
|
||||
.bus_dph_err_i (core_dph_err_i),
|
||||
.bus_haddr_i (core_haddr_i),
|
||||
.bus_hsize_i (core_hsize_i),
|
||||
.bus_priv_i (core_priv_i),
|
||||
.bus_rdata_i (core_rdata_i),
|
||||
|
||||
.bus_aph_req_d (core_aph_req_d),
|
||||
.bus_aph_excl_d (core_aph_excl_d),
|
||||
.bus_aph_ready_d (core_aph_ready_d),
|
||||
.bus_dph_ready_d (core_dph_ready_d),
|
||||
.bus_dph_err_d (core_dph_err_d),
|
||||
.bus_dph_exokay_d (core_dph_exokay_d),
|
||||
.bus_haddr_d (core_haddr_d),
|
||||
.bus_hsize_d (core_hsize_d),
|
||||
.bus_priv_d (core_priv_d),
|
||||
.bus_hwrite_d (core_hwrite_d),
|
||||
.bus_wdata_d (core_wdata_d),
|
||||
.bus_rdata_d (core_rdata_d),
|
||||
|
||||
.fence_i_vld (fence_i_vld),
|
||||
.fence_d_vld (fence_d_vld),
|
||||
.fence_rdy (fence_rdy),
|
||||
|
||||
.dbg_req_halt (dbg_req_halt),
|
||||
.dbg_req_halt_on_reset (dbg_req_halt_on_reset),
|
||||
.dbg_req_resume (dbg_req_resume),
|
||||
.dbg_halted (dbg_halted),
|
||||
.dbg_running (dbg_running),
|
||||
.dbg_data0_rdata (dbg_data0_rdata),
|
||||
.dbg_data0_wdata (dbg_data0_wdata),
|
||||
.dbg_data0_wen (dbg_data0_wen),
|
||||
.dbg_instr_data (dbg_instr_data),
|
||||
.dbg_instr_data_vld (dbg_instr_data_vld),
|
||||
.dbg_instr_data_rdy (dbg_instr_data_rdy),
|
||||
.dbg_instr_caught_exception (dbg_instr_caught_exception),
|
||||
.dbg_instr_caught_ebreak (dbg_instr_caught_ebreak),
|
||||
|
||||
.mhartid_val (mhartid_val),
|
||||
.eco_version (eco_version),
|
||||
|
||||
.irq (irq),
|
||||
.soft_irq (soft_irq),
|
||||
.timer_irq (timer_irq)
|
||||
);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Instruction port
|
||||
|
||||
localparam HTRANS_IDLE = 2'b00;
|
||||
localparam HTRANS_NSEQ = 2'b10;
|
||||
|
||||
assign i_haddr = core_haddr_i;
|
||||
assign i_htrans = core_aph_req_i ? HTRANS_NSEQ : HTRANS_IDLE;
|
||||
assign i_hsize = core_hsize_i;
|
||||
|
||||
reg dphase_active_i;
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
dphase_active_i <= 1'b0;
|
||||
end else if (i_hready) begin
|
||||
dphase_active_i <= core_aph_req_i;
|
||||
end
|
||||
end
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
// Wake->sleep transition must wait for outstanding instruction fetches to
|
||||
// complete, in particular because the arbiter clock will stop
|
||||
always @ (posedge clk) if (!rst_n) assert(clk_en || !(core_aph_req_i || dphase_active_i));
|
||||
`endif
|
||||
|
||||
assign core_aph_ready_i = i_hready && core_aph_req_i;
|
||||
assign core_dph_ready_i = i_hready && dphase_active_i;
|
||||
assign core_dph_err_i = i_hready && dphase_active_i && i_hresp;
|
||||
|
||||
assign core_rdata_i = i_hrdata;
|
||||
|
||||
assign i_hwrite = 1'b0;
|
||||
assign i_hburst = 3'h0;
|
||||
assign i_hmastlock = 1'b0;
|
||||
assign i_hmaster = 8'h00;
|
||||
assign i_hwdata = {W_DATA{1'b0}};
|
||||
|
||||
assign i_hprot = {
|
||||
2'b00, // Noncacheable/nonbufferable
|
||||
core_priv_i, // Privileged or Normal as per core state
|
||||
1'b0 // Instruction access
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Load/store port
|
||||
|
||||
// The debug module has optional System Bus Access support, which can be muxed
|
||||
// into the processor's D port here (or connected to a standalone AHB shim).
|
||||
// This confers absolutely no advantage for debugger bus throughput, but
|
||||
// allows the debugger to access the bus with minimal disturbance to the
|
||||
// processor.
|
||||
|
||||
wire bus_gnt_d;
|
||||
wire bus_gnt_s;
|
||||
|
||||
reg bus_hold_aph;
|
||||
reg [1:0] bus_gnt_ds_prev;
|
||||
reg bus_active_dph_d;
|
||||
reg bus_active_dph_s;
|
||||
|
||||
// clk_always_on is used because SBA may access the bus through this arbiter
|
||||
// whilst the core is asleep (same is not true for I-side interface)
|
||||
|
||||
always @ (posedge clk_always_on or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
bus_hold_aph <= 1'b0;
|
||||
bus_gnt_ds_prev <= 2'h0;
|
||||
end else begin
|
||||
bus_hold_aph <= d_htrans[1] && !d_hready && !d_hresp;
|
||||
bus_gnt_ds_prev <= {bus_gnt_d, bus_gnt_s};
|
||||
end
|
||||
end
|
||||
|
||||
assign {bus_gnt_d, bus_gnt_s} =
|
||||
bus_hold_aph ? bus_gnt_ds_prev :
|
||||
core_aph_req_d ? 2'b10 :
|
||||
dbg_sbus_vld && !bus_active_dph_s ? 2'b01 :
|
||||
2'b00 ;
|
||||
|
||||
always @ (posedge clk_always_on or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
bus_active_dph_d <= 1'b0;
|
||||
bus_active_dph_s <= 1'b0;
|
||||
end else if (d_hready) begin
|
||||
bus_active_dph_d <= bus_gnt_d;
|
||||
bus_active_dph_s <= bus_gnt_s;
|
||||
end
|
||||
end
|
||||
|
||||
assign d_htrans = bus_gnt_d || bus_gnt_s ? HTRANS_NSEQ : HTRANS_IDLE;
|
||||
|
||||
assign d_haddr = bus_gnt_s ? dbg_sbus_addr : core_haddr_d;
|
||||
assign d_hwrite = bus_gnt_s ? dbg_sbus_write : core_hwrite_d;
|
||||
assign d_hsize = bus_gnt_s ? {1'b0, dbg_sbus_size} : core_hsize_d;
|
||||
assign d_hexcl = bus_gnt_s ? 1'b0 : core_aph_excl_d;
|
||||
|
||||
assign d_hprot = {
|
||||
2'b00, // Noncacheable/nonbufferable
|
||||
bus_gnt_s || core_priv_d, // Privileged or Normal as per core state
|
||||
1'b1 // Data access
|
||||
};
|
||||
|
||||
assign d_hwdata = bus_active_dph_s ? dbg_sbus_wdata : core_wdata_d;
|
||||
|
||||
// D-side errors are reported even when not ready, so that the core can make
|
||||
// use of the two-phase error response to cleanly squash a second load/store
|
||||
// chasing the faulting one down the pipeline.
|
||||
assign core_aph_ready_d = d_hready && bus_gnt_d;
|
||||
assign core_dph_ready_d = bus_active_dph_d && d_hready;
|
||||
assign core_dph_err_d = bus_active_dph_d && d_hresp;
|
||||
assign core_dph_exokay_d = bus_active_dph_d && d_hexokay;
|
||||
assign core_rdata_d = d_hrdata;
|
||||
|
||||
assign dbg_sbus_err = bus_active_dph_s && d_hresp;
|
||||
assign dbg_sbus_rdy = bus_active_dph_s && d_hready;
|
||||
assign dbg_sbus_rdata = d_hrdata;
|
||||
|
||||
assign d_hburst = 3'h0;
|
||||
assign d_hmastlock = 1'b0;
|
||||
assign d_hmaster = bus_gnt_s ? 8'h01 : 8'h00;
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
Vendored
+1642
File diff suppressed because it is too large
Load Diff
+204
@@ -0,0 +1,204 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// List of addresses for CSRs implemented by Hazard3, including custom CSRs.
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// M-mode CSRs
|
||||
|
||||
// Machine Information Registers (RO)
|
||||
localparam MVENDORID = 12'hf11; // Vendor ID.
|
||||
localparam MARCHID = 12'hf12; // Architecture ID.
|
||||
localparam MIMPID = 12'hf13; // Implementation ID.
|
||||
localparam MHARTID = 12'hf14; // Hardware thread ID.
|
||||
localparam MCONFIGPTR = 12'hf15; // Pointer to configuration data structure.
|
||||
|
||||
// Machine Trap Setup (RW)
|
||||
localparam MSTATUS = 12'h300; // Machine status register.
|
||||
localparam MSTATUSH = 12'h310; // As of priv-1.12 this must be present even if tied 0.
|
||||
localparam MISA = 12'h301; // ISA and extensions
|
||||
localparam MEDELEG = 12'h302; // Machine exception delegation register.
|
||||
localparam MIDELEG = 12'h303; // Machine interrupt delegation register.
|
||||
localparam MIE = 12'h304; // Machine interrupt-enable register.
|
||||
localparam MTVEC = 12'h305; // Machine trap-handler base address.
|
||||
localparam MCOUNTEREN = 12'h306; // Machine counter enable.
|
||||
|
||||
// Machine Trap Handling (RW)
|
||||
localparam MSCRATCH = 12'h340; // Scratch register for machine trap handlers.
|
||||
localparam MEPC = 12'h341; // Machine exception program counter.
|
||||
localparam MCAUSE = 12'h342; // Machine trap cause.
|
||||
localparam MTVAL = 12'h343; // Machine bad address or instruction.
|
||||
localparam MIP = 12'h344; // Machine interrupt pending.
|
||||
|
||||
// Machine Memory Protection (RW)
|
||||
localparam PMPCFG0 = 12'h3a0; // Physical memory protection configuration.
|
||||
localparam PMPCFG1 = 12'h3a1; // Physical memory protection configuration, RV32 only.
|
||||
localparam PMPCFG2 = 12'h3a2; // Physical memory protection configuration.
|
||||
localparam PMPCFG3 = 12'h3a3; // Physical memory protection configuration, RV32 only.
|
||||
localparam PMPADDR0 = 12'h3b0; // Physical memory protection address register.
|
||||
localparam PMPADDR1 = 12'h3b1; // ...
|
||||
localparam PMPADDR2 = 12'h3b2;
|
||||
localparam PMPADDR3 = 12'h3b3;
|
||||
localparam PMPADDR4 = 12'h3b4;
|
||||
localparam PMPADDR5 = 12'h3b5;
|
||||
localparam PMPADDR6 = 12'h3b6;
|
||||
localparam PMPADDR7 = 12'h3b7;
|
||||
localparam PMPADDR8 = 12'h3b8;
|
||||
localparam PMPADDR9 = 12'h3b9;
|
||||
localparam PMPADDR10 = 12'h3ba;
|
||||
localparam PMPADDR11 = 12'h3bb;
|
||||
localparam PMPADDR12 = 12'h3bc;
|
||||
localparam PMPADDR13 = 12'h3bd;
|
||||
localparam PMPADDR14 = 12'h3be;
|
||||
localparam PMPADDR15 = 12'h3bf;
|
||||
|
||||
localparam MSECCFG = 12'h747;
|
||||
localparam MSECCFGH = 12'h757;
|
||||
|
||||
// Performance counters (RW)
|
||||
localparam MCYCLE = 12'hb00; // Raw cycles since start of day
|
||||
localparam MINSTRET = 12'hb02; // Instruction retire count since start of day
|
||||
localparam MHPMCOUNTER3 = 12'hb03; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER4 = 12'hb04; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER5 = 12'hb05; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER6 = 12'hb06; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER7 = 12'hb07; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER8 = 12'hb08; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER9 = 12'hb09; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER10 = 12'hb0a; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER11 = 12'hb0b; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER12 = 12'hb0c; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER13 = 12'hb0d; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER14 = 12'hb0e; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER15 = 12'hb0f; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER16 = 12'hb10; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER17 = 12'hb11; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER18 = 12'hb12; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER19 = 12'hb13; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER20 = 12'hb14; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER21 = 12'hb15; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER22 = 12'hb16; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER23 = 12'hb17; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER24 = 12'hb18; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER25 = 12'hb19; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER26 = 12'hb1a; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER27 = 12'hb1b; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER28 = 12'hb1c; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER29 = 12'hb1d; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER30 = 12'hb1e; // WARL (we tie to 0)
|
||||
localparam MHPMCOUNTER31 = 12'hb1f; // WARL (we tie to 0)
|
||||
|
||||
localparam MCYCLEH = 12'hb80; // High halves of each counter
|
||||
localparam MINSTRETH = 12'hb82;
|
||||
localparam MHPMCOUNTER3H = 12'hb83;
|
||||
localparam MHPMCOUNTER4H = 12'hb84;
|
||||
localparam MHPMCOUNTER5H = 12'hb85;
|
||||
localparam MHPMCOUNTER6H = 12'hb86;
|
||||
localparam MHPMCOUNTER7H = 12'hb87;
|
||||
localparam MHPMCOUNTER8H = 12'hb88;
|
||||
localparam MHPMCOUNTER9H = 12'hb89;
|
||||
localparam MHPMCOUNTER10H = 12'hb8a;
|
||||
localparam MHPMCOUNTER11H = 12'hb8b;
|
||||
localparam MHPMCOUNTER12H = 12'hb8c;
|
||||
localparam MHPMCOUNTER13H = 12'hb8d;
|
||||
localparam MHPMCOUNTER14H = 12'hb8e;
|
||||
localparam MHPMCOUNTER15H = 12'hb8f;
|
||||
localparam MHPMCOUNTER16H = 12'hb90;
|
||||
localparam MHPMCOUNTER17H = 12'hb91;
|
||||
localparam MHPMCOUNTER18H = 12'hb92;
|
||||
localparam MHPMCOUNTER19H = 12'hb93;
|
||||
localparam MHPMCOUNTER20H = 12'hb94;
|
||||
localparam MHPMCOUNTER21H = 12'hb95;
|
||||
localparam MHPMCOUNTER22H = 12'hb96;
|
||||
localparam MHPMCOUNTER23H = 12'hb97;
|
||||
localparam MHPMCOUNTER24H = 12'hb98;
|
||||
localparam MHPMCOUNTER25H = 12'hb99;
|
||||
localparam MHPMCOUNTER26H = 12'hb9a;
|
||||
localparam MHPMCOUNTER27H = 12'hb9b;
|
||||
localparam MHPMCOUNTER28H = 12'hb9c;
|
||||
localparam MHPMCOUNTER29H = 12'hb9d;
|
||||
localparam MHPMCOUNTER30H = 12'hb9e;
|
||||
localparam MHPMCOUNTER31H = 12'hb9f;
|
||||
|
||||
localparam MCOUNTINHIBIT = 12'h320; // Count inhibit register for mcycle/minstret
|
||||
localparam MHPMEVENT3 = 12'h323; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT4 = 12'h324; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT5 = 12'h325; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT6 = 12'h326; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT7 = 12'h327; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT8 = 12'h328; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT9 = 12'h329; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT10 = 12'h32a; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT11 = 12'h32b; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT12 = 12'h32c; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT13 = 12'h32d; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT14 = 12'h32e; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT15 = 12'h32f; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT16 = 12'h330; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT17 = 12'h331; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT18 = 12'h332; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT19 = 12'h333; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT20 = 12'h334; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT21 = 12'h335; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT22 = 12'h336; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT23 = 12'h337; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT24 = 12'h338; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT25 = 12'h339; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT26 = 12'h33a; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT27 = 12'h33b; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT28 = 12'h33c; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT29 = 12'h33d; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT30 = 12'h33e; // WARL (we tie to 0)
|
||||
localparam MHPMEVENT31 = 12'h33f; // WARL (we tie to 0)
|
||||
|
||||
// Other standard M-mode CSRs:
|
||||
localparam MENVCFG = 12'h30a;
|
||||
localparam MENVCFGH = 12'h31a;
|
||||
|
||||
// Custom M-mode CSRs:
|
||||
localparam PMPCFGM0 = 12'hbd0; // Make PMP regions M-mode without locking
|
||||
// bd1 // (reserved for >32 regions)
|
||||
|
||||
localparam MEIEA = 12'hbe0; // External interrupt pending array
|
||||
localparam MEIPA = 12'hbe1; // External interrupt enable array
|
||||
localparam MEIFA = 12'hbe2; // External interrupt force array
|
||||
localparam MEIPRA = 12'hbe3; // External interrupt priority array
|
||||
localparam MEINEXT = 12'hbe4; // Next external interrupt
|
||||
localparam MEICONTEXT = 12'hbe5; // External interrupt context register
|
||||
|
||||
localparam MSLEEP = 12'hbf0; // M-mode sleep control register
|
||||
localparam H3MISA = 12'hbf1; // Hazard3 M-mode ISA identification register
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// U-mode CSRs
|
||||
|
||||
// Read-only aliases of M-mode counter CSRs:
|
||||
localparam CYCLE = 12'hc00;
|
||||
localparam TIME = 12'hc01;
|
||||
localparam INSTRET = 12'hc02;
|
||||
localparam CYCLEH = 12'hc80;
|
||||
localparam TIMEH = 12'hc81;
|
||||
localparam INSTRETH = 12'hc82;
|
||||
|
||||
// Custom U-mode CSRs
|
||||
localparam SLEEP = 12'h8f0; // U-mode subset of M-mode sleep control
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Trigger Module
|
||||
|
||||
localparam TSELECT = 12'h7a0;
|
||||
localparam TDATA1 = 12'h7a1;
|
||||
localparam TDATA2 = 12'h7a2;
|
||||
localparam TDATA3 = 12'h7a3;
|
||||
localparam TINFO = 12'h7a4;
|
||||
localparam TCONTROL = 12'h7a5;
|
||||
localparam MCONTEXT = 12'h7a8;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// D-mode CSRs
|
||||
|
||||
localparam DCSR = 12'h7b0;
|
||||
localparam DPC = 12'h7b1;
|
||||
localparam DMDATA0 = 12'hbff; // Custom read/write
|
||||
Vendored
+594
@@ -0,0 +1,594 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2023 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_decode #(
|
||||
`include "hazard3_config.vh"
|
||||
,
|
||||
`include "hazard3_width_const.vh"
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
|
||||
input wire [31:0] fd_cir,
|
||||
input wire [1:0] fd_cir_err,
|
||||
input wire [1:0] fd_cir_predbranch,
|
||||
input wire [1:0] fd_cir_vld,
|
||||
input wire fd_cir_is_32bit,
|
||||
input wire fd_cir_invalid_16bit,
|
||||
input wire fd_cir_is_uop,
|
||||
input wire fd_cir_uop_nonfinal,
|
||||
input wire fd_cir_uop_no_pc_update,
|
||||
input wire fd_cir_uop_atomic,
|
||||
|
||||
output wire [1:0] df_cir_use,
|
||||
output wire df_cir_flush_behind,
|
||||
|
||||
output wire df_uop_stall,
|
||||
output wire df_uop_clear,
|
||||
output wire df_lspair_phase_next,
|
||||
|
||||
output wire [W_ADDR-1:0] d_pc,
|
||||
|
||||
input wire debug_mode,
|
||||
input wire m_mode,
|
||||
input wire trap_wfi,
|
||||
|
||||
input wire [W_ADDR-1:0] debug_dpc_wdata,
|
||||
input wire debug_dpc_wen,
|
||||
output wire [W_ADDR-1:0] debug_dpc_rdata,
|
||||
|
||||
output wire d_starved,
|
||||
input wire x_stall,
|
||||
input wire f_jump_now,
|
||||
input wire [W_ADDR-1:0] f_jump_target,
|
||||
input wire x_jump_not_except,
|
||||
input wire [W_ADDR-1:0] d_btb_target_addr,
|
||||
|
||||
output reg [W_DATA-1:0] d_imm,
|
||||
output reg [W_REGADDR-1:0] d_rs1,
|
||||
output reg [W_REGADDR-1:0] d_rs2,
|
||||
output reg [W_REGADDR-1:0] d_rd,
|
||||
output reg [2:0] d_funct3_32b,
|
||||
output reg [6:0] d_funct7_32b,
|
||||
output reg [W_ALUSRC-1:0] d_alusrc_a,
|
||||
output reg [W_ALUSRC-1:0] d_alusrc_b,
|
||||
output reg [W_ALUOP-1:0] d_aluop,
|
||||
output reg [W_MEMOP-1:0] d_memop,
|
||||
output reg [W_MULOP-1:0] d_mulop,
|
||||
output reg d_csr_ren,
|
||||
output reg d_csr_wen,
|
||||
output reg [1:0] d_csr_wtype,
|
||||
output reg d_csr_w_imm,
|
||||
output reg [W_BCOND-1:0] d_branchcond,
|
||||
output reg [W_ADDR-1:0] d_addr_offs,
|
||||
output reg d_addr_is_regoffs,
|
||||
output reg [W_EXCEPT-1:0] d_except,
|
||||
output reg d_sleep_wfi,
|
||||
output reg d_sleep_block,
|
||||
output reg d_sleep_unblock,
|
||||
output wire d_no_pc_increment,
|
||||
output wire d_uninterruptible,
|
||||
output wire [W_ADDR-1:0] d_lspair_offset,
|
||||
output reg d_fence_i,
|
||||
output reg d_fence_d
|
||||
);
|
||||
|
||||
`include "rv_opcodes.vh"
|
||||
`include "hazard3_ops.vh"
|
||||
|
||||
localparam HAVE_CSR = CSR_M_MANDATORY || CSR_M_TRAP || CSR_COUNTER;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
wire [31:0] d_instr = fd_cir | {
|
||||
30'd0, {2{~|EXTENSION_C}}
|
||||
};
|
||||
|
||||
reg d_invalid_32bit;
|
||||
wire d_invalid = fd_cir_invalid_16bit || d_invalid_32bit;
|
||||
|
||||
assign d_uninterruptible = |EXTENSION_ZCMP && fd_cir_uop_atomic;
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
always @ (posedge clk) if (rst_n) begin
|
||||
assert(!(d_invalid && fd_cir_is_uop));
|
||||
assert(!(d_invalid && fd_cir_uop_atomic));
|
||||
end
|
||||
`endif
|
||||
|
||||
wire d_lspair_nonfinal;
|
||||
// Signal to null the mepc offset when taking an exception on this
|
||||
// instruction (because uops in a sequence *which can except*, so excluding
|
||||
// the final sp adjust on popret/popretz, will all have the same PC as the
|
||||
// next uop, which will be in stage 2 when they take their exception)
|
||||
assign d_no_pc_increment = fd_cir_uop_nonfinal || d_lspair_nonfinal;
|
||||
|
||||
assign df_uop_stall = x_stall || d_starved;
|
||||
|
||||
// Note !df_cir_flush_behind because the jump in cm.popret/popretz is the
|
||||
// *penultimate* instruction: we execute the stack adjustment in the fetch
|
||||
// bubble to save a cycle, still need to finish the uop sequence.
|
||||
//
|
||||
// The sp adjust cannot generate an exception (it's an `add` with the same
|
||||
// PMP.X and breakpoint comparison results as earlier uops) and interrupts are
|
||||
// suppressed for this part of the sequence.
|
||||
assign df_uop_clear = f_jump_now && !df_cir_flush_behind;
|
||||
|
||||
// Decode various immediate formats
|
||||
wire [31:0] d_imm_i = {{21{d_instr[31]}}, d_instr[30:20]};
|
||||
wire [31:0] d_imm_s = {{21{d_instr[31]}}, d_instr[30:25], d_instr[11:7]};
|
||||
wire [31:0] d_imm_b = {{20{d_instr[31]}}, d_instr[7], d_instr[30:25], d_instr[11:8], 1'b0};
|
||||
wire [31:0] d_imm_u = {d_instr[31:12], {12{1'b0}}};
|
||||
wire [31:0] d_imm_j = {{12{d_instr[31]}}, d_instr[19:12], d_instr[20], d_instr[30:21], 1'b0};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// PC/CIR control
|
||||
|
||||
// Must not flag bus error for a valid 16-bit instruction *followed by* an
|
||||
// error, because instruction fetch errors are speculative, and can be
|
||||
// flushed by e.g. a branch instruction. Note the 16 LSBs must be valid for
|
||||
// us to know an instruction's size.
|
||||
wire d_except_instr_bus_fault = fd_cir_vld > 2'd0 && fd_cir_err[0] ||
|
||||
fd_cir_vld > 2'd1 && fd_cir_is_32bit && fd_cir_err[1];
|
||||
|
||||
assign d_starved = ~|fd_cir_vld || fd_cir_vld[0] && fd_cir_is_32bit;
|
||||
|
||||
wire d_stall = x_stall || d_starved || fd_cir_uop_nonfinal || d_lspair_nonfinal;
|
||||
|
||||
assign df_cir_use =
|
||||
d_starved || d_stall ? 2'h0 :
|
||||
fd_cir_is_32bit ? 2'h2 : 2'h1;
|
||||
|
||||
// CIR Locking is required if we successfully assert a jump request, but
|
||||
// decode is stalled. It is not possible to gate the jump request if the
|
||||
// stall depends on bus stall (as this would create a through-path from bus
|
||||
// stall to bus request) so instead we instruct the frontend to preserve the
|
||||
// stalled instruction when flushing, and fill in behind it.
|
||||
//
|
||||
// Once the stall clears, the stalled instruction can execute its remaining
|
||||
// side effects e.g. writing a link value to the register file.
|
||||
wire jump_caused_by_d = f_jump_now && x_jump_not_except;
|
||||
wire assert_cir_lock = jump_caused_by_d && d_stall;
|
||||
|
||||
// CIR lock ends naturally when an instruction (not just uop) graduates to the
|
||||
// next stage:
|
||||
wire finished_cir_lock = !d_stall;
|
||||
|
||||
// CIR lock can meet an untimely end due to trap entry. One way to reach this
|
||||
// is a dphase load fault on the final load in a cm.popret: here the `ret`
|
||||
// issues a fetch address while stalled on the first dphase cycle, then is
|
||||
// flushed by trap on second cycle.
|
||||
wire deassert_cir_lock = finished_cir_lock || (f_jump_now && !x_jump_not_except);
|
||||
|
||||
reg cir_lock_prev;
|
||||
wire cir_lock = (cir_lock_prev && !deassert_cir_lock) || assert_cir_lock;
|
||||
assign df_cir_flush_behind = assert_cir_lock && !cir_lock_prev;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
cir_lock_prev <= 1'b0;
|
||||
end else begin
|
||||
cir_lock_prev <= cir_lock;
|
||||
end
|
||||
end
|
||||
|
||||
reg [W_ADDR-1:0] pc;
|
||||
wire [W_ADDR-1:0] pc_seq_next = pc + (
|
||||
|EXTENSION_ZCMP && fd_cir_is_uop && fd_cir_uop_no_pc_update ? 32'd0 :
|
||||
fd_cir_is_32bit ? 32'd4 : 32'd2
|
||||
);
|
||||
|
||||
assign d_pc = pc;
|
||||
assign debug_dpc_rdata = pc;
|
||||
|
||||
// Frontend should mark the whole instruction, and nothing but the
|
||||
// instruction, as a predicted branch. This goes wrong when we execute the
|
||||
// address containing the predicted branch twice with different 16-bit
|
||||
// alignments (!). We need to issue a branch-to-self to get back on a linear
|
||||
// path, otherwise PC and CIR will diverge and we will misexecute.
|
||||
wire partial_predicted_branch = !d_starved &&
|
||||
|BRANCH_PREDICTOR && fd_cir_is_32bit && ^fd_cir_predbranch;
|
||||
|
||||
wire predicted_branch = |BRANCH_PREDICTOR && fd_cir_predbranch[0];
|
||||
|
||||
// Generally locking takes place on a stalled jump/branch, which may need the
|
||||
// original PC available to produce a link address when it unstalls. An
|
||||
// exception to this is jumps in micro-op sequences: in this case the jump is
|
||||
// the penultimate instruction in the sequence (ret before addi sp) and we
|
||||
// need to capture the pc mid-uop-sequence.
|
||||
wire hold_pc_on_cir_lock = assert_cir_lock && !(fd_cir_is_uop && !fd_cir_uop_no_pc_update && !x_stall);
|
||||
wire update_pc_on_cir_unlock = cir_lock_prev && finished_cir_lock && !fd_cir_uop_no_pc_update;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
pc <= RESET_VECTOR;
|
||||
end else begin
|
||||
if (debug_dpc_wen) begin
|
||||
pc <= debug_dpc_wdata;
|
||||
end else if (debug_mode) begin
|
||||
pc <= pc;
|
||||
end else if ((f_jump_now && !hold_pc_on_cir_lock) || update_pc_on_cir_unlock) begin
|
||||
pc <= f_jump_target;
|
||||
end else if (!f_jump_now && fd_cir_uop_nonfinal && !fd_cir_uop_no_pc_update && !x_stall) begin
|
||||
// End of previously stalled jr uop in cm.popret and cm.popretz:
|
||||
// safe to update PC as next instruction (addi sp) cannot trap.
|
||||
pc <= f_jump_target;
|
||||
end else if (!d_stall && !cir_lock) begin
|
||||
// If this instruction is a predicted-taken branch (and has not
|
||||
// generated a mispredict recovery jump) then set PC to the
|
||||
// prediction target instead of the sequentially next PC
|
||||
pc <= predicted_branch ? d_btb_target_addr : pc_seq_next;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
always @ (posedge clk) if (rst_n) begin
|
||||
if (~|fd_cir_vld) assert(!fd_cir_is_uop);
|
||||
if (fd_cir_uop_no_pc_update) assert(fd_cir_is_uop);
|
||||
if (fd_cir_uop_nonfinal) assert(fd_cir_is_uop);
|
||||
if ($past(df_uop_clear)) assert(!fd_cir_is_uop);
|
||||
// Important to avoid spurious PC updates following a trap on the final
|
||||
// load of a cm.popret:
|
||||
if ($past(df_uop_clear)) assert(!fd_cir_uop_no_pc_update);
|
||||
end
|
||||
`endif
|
||||
|
||||
wire [W_ADDR-1:0] branch_offs =
|
||||
!fd_cir_is_32bit && predicted_branch ? 32'd2 :
|
||||
fd_cir_is_32bit && predicted_branch ? 32'd4 : d_imm_b;
|
||||
|
||||
always @ (*) begin
|
||||
casez ({|EXTENSION_A, d_instr[6:2]})
|
||||
{1'bz, 5'b11011}: d_addr_offs = d_imm_j ; // JAL
|
||||
{1'bz, 5'b11000}: d_addr_offs = branch_offs ; // Branches
|
||||
{1'bz, 5'b01000}: d_addr_offs = d_imm_s ; // Store
|
||||
{1'bz, 5'b11001}: d_addr_offs = d_imm_i ; // JALR
|
||||
{1'bz, 5'b00000}: d_addr_offs = d_imm_i ; // Loads
|
||||
{1'b1, 5'b01011}: d_addr_offs = 32'h0000_0000; // Atomics
|
||||
default: d_addr_offs = 32'hxxxx_xxxx;
|
||||
endcase
|
||||
if (partial_predicted_branch) begin
|
||||
d_addr_offs = 32'h0000_0000;
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Track phase of load/store pair instructions (Zilsd and Zclsd)
|
||||
|
||||
// This could be shared with uop_ctr (for Zcmp) but the two are fundamentally
|
||||
// different: Zcmp has 16-bit instructions which expand to sequences of
|
||||
// 32-bit, whereas Zilsd has multi-phase 32-bit instructions and Zclsd has
|
||||
// direct 16-bit aliases of those instructions. Therefore it's cleaner to
|
||||
// separate the phasing from the decompression for Zilsd/Zclsd.
|
||||
|
||||
wire d_lspair_phase;
|
||||
|
||||
// Reorder accesses to avoid clobbering rs1 (base) in first half of load:
|
||||
wire d_lspair_reg_sel = d_lspair_phase == d_instr[15];
|
||||
|
||||
generate
|
||||
if (EXTENSION_ZILSD) begin: have_lspair_reg_sel
|
||||
reg d_lspair_phase_r;
|
||||
assign d_lspair_phase = d_lspair_phase_r;
|
||||
reg instr_is_lspair;
|
||||
always @ (*) begin
|
||||
casez ({d_invalid || d_starved, d_instr})
|
||||
{1'b0, `RVOPC_LD}: instr_is_lspair = 1'b1;
|
||||
{1'b0, `RVOPC_SD}: instr_is_lspair = 1'b1;
|
||||
default: instr_is_lspair = 1'b0;
|
||||
endcase
|
||||
end
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
d_lspair_phase_r <= 1'b0;
|
||||
end else begin
|
||||
d_lspair_phase_r <= df_lspair_phase_next;
|
||||
end
|
||||
end
|
||||
|
||||
assign df_lspair_phase_next =
|
||||
!d_stall || f_jump_now ? 1'b0 :
|
||||
instr_is_lspair && !x_stall ? 1'b1 : d_lspair_phase_r;
|
||||
|
||||
assign d_lspair_nonfinal = instr_is_lspair && !d_lspair_phase_r;
|
||||
|
||||
assign d_lspair_offset = {
|
||||
29'h0,
|
||||
d_lspair_reg_sel && instr_is_lspair,
|
||||
2'h0
|
||||
};
|
||||
|
||||
end else begin: no_lspair_reg_sel
|
||||
|
||||
assign d_lspair_phase = 1'b0;
|
||||
assign df_lspair_phase_next = 1'b0;
|
||||
assign d_lspair_nonfinal = 1'b0;
|
||||
assign d_lspair_offset = 32'd0;
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Decode X controls
|
||||
|
||||
localparam X0 = {W_REGADDR{1'b0}};
|
||||
|
||||
// First decode the instruction bits, based on available extensions and
|
||||
// privilege state, without gating in any stall/exception signals.
|
||||
reg [W_REGADDR-1:0] raw_rs1;
|
||||
reg [W_REGADDR-1:0] raw_rs2;
|
||||
reg [W_REGADDR-1:0] raw_rd;
|
||||
reg [W_DATA-1:0] raw_imm;
|
||||
reg [W_ALUSRC-1:0] raw_alusrc_a;
|
||||
reg [W_ALUSRC-1:0] raw_alusrc_b;
|
||||
reg [W_ALUOP-1:0] raw_aluop;
|
||||
reg [W_MEMOP-1:0] raw_memop;
|
||||
reg [W_MULOP-1:0] raw_mulop;
|
||||
reg raw_csr_ren;
|
||||
reg raw_csr_wen;
|
||||
reg [1:0] raw_csr_wtype;
|
||||
reg raw_csr_w_imm;
|
||||
reg [W_BCOND-1:0] raw_branchcond;
|
||||
reg raw_addr_is_regoffs;
|
||||
reg [W_EXCEPT-1:0] raw_except;
|
||||
reg raw_sleep_wfi;
|
||||
reg raw_sleep_block;
|
||||
reg raw_sleep_unblock;
|
||||
reg raw_fence_i;
|
||||
reg raw_fence_d;
|
||||
|
||||
always @ (*) begin
|
||||
// Assign some defaults
|
||||
raw_rs1 = d_instr[19:15];
|
||||
raw_rs2 = d_instr[24:20];
|
||||
raw_rd = d_instr[11: 7];
|
||||
raw_imm = d_imm_i;
|
||||
raw_alusrc_a = ALUSRCA_RS1;
|
||||
raw_alusrc_b = ALUSRCB_RS2;
|
||||
raw_aluop = ALUOP_ADD;
|
||||
raw_memop = MEMOP_NONE;
|
||||
raw_mulop = M_OP_MUL;
|
||||
raw_csr_ren = 1'b0;
|
||||
raw_csr_wen = 1'b0;
|
||||
raw_csr_wtype = CSR_WTYPE_W;
|
||||
raw_csr_w_imm = 1'b0;
|
||||
raw_branchcond = BCOND_NEVER;
|
||||
raw_addr_is_regoffs = 1'b0;
|
||||
raw_except = EXCEPT_NONE;
|
||||
raw_sleep_wfi = 1'b0;
|
||||
raw_sleep_block = 1'b0;
|
||||
raw_sleep_unblock = 1'b0;
|
||||
raw_fence_i = 1'b0;
|
||||
raw_fence_d = 1'b0;
|
||||
// Note this funct3/funct7 are valid only for 32-bit instructions. They
|
||||
// are useful for clusters of related ALU ops, such as sh*add, clmul.
|
||||
d_funct3_32b = fd_cir[14:12];
|
||||
d_funct7_32b = fd_cir[31:25];
|
||||
|
||||
d_invalid_32bit = 1'b0;
|
||||
|
||||
casez (d_instr)
|
||||
`RVOPC_BEQ: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_SUB; raw_branchcond = BCOND_ZERO; end
|
||||
`RVOPC_BNE: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_SUB; raw_branchcond = BCOND_NZERO; end
|
||||
`RVOPC_BLT: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_LT; raw_branchcond = BCOND_NZERO; end
|
||||
`RVOPC_BGE: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_LT; raw_branchcond = BCOND_ZERO; end
|
||||
`RVOPC_BLTU: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_LTU; raw_branchcond = BCOND_NZERO; end
|
||||
`RVOPC_BGEU: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_rd = X0; raw_aluop = ALUOP_LTU; raw_branchcond = BCOND_ZERO; end
|
||||
`RVOPC_JALR: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_branchcond = BCOND_ALWAYS; raw_addr_is_regoffs = 1'b1;
|
||||
raw_rs2 = X0; raw_aluop = ALUOP_ADD; raw_alusrc_a = ALUSRCA_PC; raw_alusrc_b = ALUSRCB_IMM; raw_imm = fd_cir_is_32bit ? 32'd4 : 32'd2; end
|
||||
`RVOPC_JAL: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_branchcond = BCOND_ALWAYS; raw_rs1 = X0;
|
||||
raw_rs2 = X0; raw_aluop = ALUOP_ADD; raw_alusrc_a = ALUSRCA_PC; raw_alusrc_b = ALUSRCB_IMM; raw_imm = fd_cir_is_32bit ? 32'd4 : 32'd2; end
|
||||
`RVOPC_LUI: begin raw_aluop = ALUOP_RS2; raw_imm = d_imm_u; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; raw_rs1 = X0; end
|
||||
`RVOPC_AUIPC: begin d_invalid_32bit = DEBUG_SUPPORT && debug_mode; raw_aluop = ALUOP_ADD; raw_imm = d_imm_u; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; raw_alusrc_a = ALUSRCA_PC; raw_rs1 = X0; end
|
||||
`RVOPC_ADDI: begin raw_aluop = ALUOP_ADD; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
|
||||
`RVOPC_SLLI: begin raw_aluop = ALUOP_SLL; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
|
||||
`RVOPC_SLTI: begin raw_aluop = ALUOP_LT; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
|
||||
`RVOPC_SLTIU: begin raw_aluop = ALUOP_LTU; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
|
||||
`RVOPC_XORI: begin raw_aluop = ALUOP_XOR; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
|
||||
`RVOPC_SRLI: begin raw_aluop = ALUOP_SRL; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
|
||||
`RVOPC_SRAI: begin raw_aluop = ALUOP_SRA; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
|
||||
`RVOPC_ORI: begin raw_aluop = ALUOP_OR; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
|
||||
`RVOPC_ANDI: begin raw_aluop = ALUOP_AND; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; raw_rs2 = X0; end
|
||||
`RVOPC_ADD: begin raw_aluop = ALUOP_ADD; end
|
||||
`RVOPC_SUB: begin raw_aluop = ALUOP_SUB; end
|
||||
`RVOPC_SLL: begin raw_aluop = ALUOP_SLL; end
|
||||
`RVOPC_SLTU: begin raw_aluop = ALUOP_LTU; end
|
||||
`RVOPC_XOR: begin raw_aluop = ALUOP_XOR; end
|
||||
`RVOPC_SRL: begin raw_aluop = ALUOP_SRL; end
|
||||
`RVOPC_SRA: begin raw_aluop = ALUOP_SRA; end
|
||||
`RVOPC_OR: begin raw_aluop = ALUOP_OR; end
|
||||
`RVOPC_AND: begin raw_aluop = ALUOP_AND; end
|
||||
`RVOPC_LB: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LB; end
|
||||
`RVOPC_LH: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LH; end
|
||||
`RVOPC_LW: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LW; end
|
||||
`RVOPC_LBU: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LBU; end
|
||||
`RVOPC_LHU: begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LHU; end
|
||||
`RVOPC_SB: begin raw_addr_is_regoffs = 1'b1; raw_aluop = ALUOP_RS2; raw_memop = MEMOP_SB; raw_rd = X0; end
|
||||
`RVOPC_SH: begin raw_addr_is_regoffs = 1'b1; raw_aluop = ALUOP_RS2; raw_memop = MEMOP_SH; raw_rd = X0; end
|
||||
`RVOPC_SW: begin raw_addr_is_regoffs = 1'b1; raw_aluop = ALUOP_RS2; raw_memop = MEMOP_SW; raw_rd = X0; end
|
||||
|
||||
`RVOPC_SLT: begin
|
||||
raw_aluop = ALUOP_LT;
|
||||
if (|EXTENSION_XH3POWER && ~|raw_rd && ~|raw_rs1) begin
|
||||
if (raw_rs2 == 5'h00) begin
|
||||
// h3.block (power management hint)
|
||||
d_invalid_32bit = trap_wfi;
|
||||
raw_sleep_block = !trap_wfi;
|
||||
end else if (raw_rs2 == 5'h01) begin
|
||||
// h3.unblock (power management hint)
|
||||
d_invalid_32bit = trap_wfi;
|
||||
raw_sleep_unblock = !trap_wfi;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
`RVOPC_MUL: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_MUL; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_MULH: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_MULH; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_MULHSU: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_MULHSU; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_MULHU: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_MULHU; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_DIV: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_DIV; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_DIVU: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_DIVU; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_REM: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_REM; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_REMU: if (EXTENSION_M) begin raw_aluop = ALUOP_MULDIV; raw_mulop = M_OP_REMU; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_LR_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_LR_W; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_SC_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_SC_W; raw_aluop = ALUOP_RS2; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_AMOSWAP_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_RS2; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_AMOADD_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_ADD; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_AMOXOR_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_XOR; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_AMOAND_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_AND; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_AMOOR_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_OR; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_AMOMIN_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_MIN; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_AMOMAX_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_MAX; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_AMOMINU_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_MINU; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_AMOMAXU_W: if (EXTENSION_A) begin raw_addr_is_regoffs = 1'b1; raw_memop = MEMOP_AMO; raw_aluop = ALUOP_MAXU; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_LD: if (EXTENSION_ZILSD) begin raw_addr_is_regoffs = 1'b1; raw_rs2 = X0; raw_memop = MEMOP_LW; raw_rd = {d_instr[11: 8], d_lspair_reg_sel}; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_SD: if (EXTENSION_ZILSD) begin raw_addr_is_regoffs = 1'b1; raw_rd = X0; raw_memop = MEMOP_SW; raw_rs2 = {d_instr[24:21], d_lspair_reg_sel}; raw_aluop = ALUOP_RS2; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_SH1ADD: if (EXTENSION_ZBA) begin raw_aluop = ALUOP_SHXADD; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_SH2ADD: if (EXTENSION_ZBA) begin raw_aluop = ALUOP_SHXADD; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_SH3ADD: if (EXTENSION_ZBA) begin raw_aluop = ALUOP_SHXADD; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_ANDN: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ANDN; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CLZ: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_CLZ; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CPOP: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_CPOP; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CTZ: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_CTZ; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_MAX: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_MAX; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_MAXU: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_MAXU; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_MIN: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_MIN; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_MINU: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_MINU; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_ORC_B: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ORC_B; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_ORN: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ORN; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_REV8: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_REV8; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_ROL: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ROL; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_ROR: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ROR; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_RORI: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ROR; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_SEXT_B: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_SEXT_B; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_SEXT_H: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_SEXT_H; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_XNOR: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_XNOR; end else begin d_invalid_32bit = 1'b1; end
|
||||
// Note: ZEXT_H is a subset of PACK from Zbkb. This is fine as long
|
||||
// as this case appears first, since Zbkb implies Zbb on Hazard3.
|
||||
`RVOPC_ZEXT_H: if (EXTENSION_ZBB) begin raw_aluop = ALUOP_ZEXT_H; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_CLMUL: if (EXTENSION_ZBC) begin raw_aluop = ALUOP_CLMUL; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CLMULH: if (EXTENSION_ZBC) begin raw_aluop = ALUOP_CLMUL; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CLMULR: if (EXTENSION_ZBC) begin raw_aluop = ALUOP_CLMUL; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_BCLR: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BCLR; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_BCLRI: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BCLR; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_BEXT: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BEXT; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_BEXTI: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BEXT; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_BINV: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BINV; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_BINVI: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BINV; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_BSET: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BSET; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_BSETI: if (EXTENSION_ZBS) begin raw_aluop = ALUOP_BSET; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_PACK: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_PACK; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_PACKH: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_PACKH; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_BREV8: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_BREV8; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_UNZIP: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_UNZIP; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_ZIP: if (EXTENSION_ZBKB) begin raw_aluop = ALUOP_ZIP; raw_rs2 = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_XPERM8: if (EXTENSION_ZBKX) begin raw_aluop = ALUOP_XPERM; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_XPERM4: if (EXTENSION_ZBKX) begin raw_aluop = ALUOP_XPERM; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_H3_BEXTM: if (EXTENSION_XH3BEXTM) begin
|
||||
raw_aluop = ALUOP_BEXTM; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_H3_BEXTMI: if (EXTENSION_XH3BEXTM) begin
|
||||
raw_aluop = ALUOP_BEXTM; raw_rs2 = X0; raw_imm = d_imm_i; raw_alusrc_b = ALUSRCB_IMM; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CSRRW: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = 1'b1 ; raw_csr_ren = |raw_rd; raw_csr_wtype = CSR_WTYPE_W; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CSRRS: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = |raw_rs1; raw_csr_ren = 1'b1 ; raw_csr_wtype = CSR_WTYPE_S; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CSRRC: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = |raw_rs1; raw_csr_ren = 1'b1 ; raw_csr_wtype = CSR_WTYPE_C; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CSRRWI: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = 1'b1 ; raw_csr_ren = |raw_rd; raw_csr_wtype = CSR_WTYPE_W; raw_csr_w_imm = 1'b1; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CSRRSI: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = |raw_rs1; raw_csr_ren = 1'b1 ; raw_csr_wtype = CSR_WTYPE_S; raw_csr_w_imm = 1'b1; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_CSRRCI: if (HAVE_CSR) begin raw_rs2 = X0; raw_imm = d_imm_i; raw_csr_wen = |raw_rs1; raw_csr_ren = 1'b1 ; raw_csr_wtype = CSR_WTYPE_C; raw_csr_w_imm = 1'b1; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
`RVOPC_FENCE: begin raw_rs2 = X0; raw_fence_d = 1'b1; end // Note rs1/rd are zero in instruction
|
||||
`RVOPC_FENCE_I: if (EXTENSION_ZIFENCEI) begin raw_except = debug_mode ? EXCEPT_NONE : EXCEPT_REFETCH; raw_fence_i = 1'b1; end else begin d_invalid_32bit = 1'b1; end // note rs1/rs2/rd are zero in instruction
|
||||
`RVOPC_ECALL: if (HAVE_CSR) begin raw_except = m_mode || !U_MODE ? EXCEPT_ECALL_M : EXCEPT_ECALL_U; raw_rs2 = X0; raw_rs1 = X0; raw_rd = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_EBREAK: if (HAVE_CSR) begin raw_except = EXCEPT_EBREAK; raw_rs2 = X0; raw_rs1 = X0; raw_rd = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_MRET: if (HAVE_CSR && m_mode) begin raw_except = EXCEPT_MRET; raw_rs2 = X0; raw_rs1 = X0; raw_rd = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
`RVOPC_WFI: if (HAVE_CSR && !trap_wfi) begin raw_sleep_wfi = 1'b1; raw_rs2 = X0; raw_rs1 = X0; raw_rd = X0; end else begin d_invalid_32bit = 1'b1; end
|
||||
|
||||
default: begin d_invalid_32bit = 1'b1; end
|
||||
endcase
|
||||
|
||||
if (|EXTENSION_E && (raw_rd[4] || raw_rs1[4] || raw_rs2[4])) begin
|
||||
d_invalid_32bit = 1'b1;
|
||||
end
|
||||
end
|
||||
|
||||
// Then gate key signals based on CIR fullness, fetch faults etc. The split
|
||||
// helps to avoid an event scheduling feedback loop that makes simulators
|
||||
// unhappy and slow, particularly verilator
|
||||
|
||||
localparam [4:0] REGADDR_MASK = {~|EXTENSION_E, 4'hf};
|
||||
|
||||
always @ (*) begin
|
||||
// Pass through by default
|
||||
d_rs1 = raw_rs1 & REGADDR_MASK;
|
||||
d_rs2 = raw_rs2 & REGADDR_MASK;
|
||||
d_rd = raw_rd & REGADDR_MASK;
|
||||
d_imm = raw_imm;
|
||||
d_alusrc_a = raw_alusrc_a;
|
||||
d_alusrc_b = raw_alusrc_b;
|
||||
d_aluop = raw_aluop;
|
||||
d_memop = raw_memop;
|
||||
d_mulop = raw_mulop;
|
||||
d_csr_ren = raw_csr_ren;
|
||||
d_csr_wen = raw_csr_wen;
|
||||
d_csr_wtype = raw_csr_wtype;
|
||||
d_csr_w_imm = raw_csr_w_imm;
|
||||
d_branchcond = raw_branchcond;
|
||||
d_addr_is_regoffs = raw_addr_is_regoffs;
|
||||
d_except = raw_except;
|
||||
d_sleep_wfi = raw_sleep_wfi;
|
||||
d_sleep_block = raw_sleep_block;
|
||||
d_sleep_unblock = raw_sleep_unblock;
|
||||
d_fence_i = raw_fence_i;
|
||||
d_fence_d = raw_fence_d;
|
||||
|
||||
if (d_invalid || d_starved || d_except_instr_bus_fault || partial_predicted_branch) begin
|
||||
d_rs1 = {W_REGADDR{1'b0}};
|
||||
d_rs2 = {W_REGADDR{1'b0}};
|
||||
d_rd = {W_REGADDR{1'b0}};
|
||||
d_memop = MEMOP_NONE;
|
||||
d_branchcond = BCOND_NEVER;
|
||||
d_csr_ren = 1'b0;
|
||||
d_csr_wen = 1'b0;
|
||||
d_except = EXCEPT_NONE;
|
||||
d_sleep_wfi = 1'b0;
|
||||
d_sleep_block = 1'b0;
|
||||
d_sleep_unblock = 1'b0;
|
||||
d_fence_i = 1'b0;
|
||||
d_fence_d = 1'b0;
|
||||
|
||||
if (EXTENSION_M)
|
||||
d_aluop = ALUOP_ADD;
|
||||
|
||||
if (d_except_instr_bus_fault)
|
||||
d_except = EXCEPT_INSTR_FAULT;
|
||||
else if (d_invalid && !d_starved)
|
||||
d_except = EXCEPT_INSTR_ILLEGAL;
|
||||
end
|
||||
if (partial_predicted_branch) begin
|
||||
d_addr_is_regoffs = 1'b0;
|
||||
d_branchcond = BCOND_ALWAYS;
|
||||
end
|
||||
if (cir_lock_prev) begin
|
||||
d_branchcond = BCOND_NEVER;
|
||||
end
|
||||
end
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+906
@@ -0,0 +1,906 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_frontend #(
|
||||
`include "hazard3_config.vh"
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
|
||||
// Fetch interface
|
||||
// addr_vld may be asserted at any time, but after assertion,
|
||||
// neither addr nor addr_vld may change until the cycle after addr_rdy.
|
||||
// There is no backpressure on the data interface; the front end
|
||||
// must ensure it does not request data it cannot receive.
|
||||
// addr_rdy and dat_vld may be functions of hready, and
|
||||
// may not be used to compute combinational outputs.
|
||||
output wire mem_size, // 1'b1 -> 32 bit access
|
||||
output wire [W_ADDR-1:0] mem_addr,
|
||||
output wire mem_priv,
|
||||
output wire mem_addr_vld,
|
||||
input wire mem_addr_rdy,
|
||||
input wire [W_DATA-1:0] mem_data,
|
||||
input wire mem_data_err,
|
||||
input wire mem_data_vld,
|
||||
|
||||
// Jump/flush interface
|
||||
// Processor may assert vld at any time. The request will not go through
|
||||
// unless rdy is high. Processor *may* alter request during this time.
|
||||
// Inputs must not be a function of hready.
|
||||
input wire [W_ADDR-1:0] jump_target,
|
||||
input wire jump_priv,
|
||||
input wire jump_target_vld,
|
||||
output wire jump_target_rdy,
|
||||
|
||||
// Interface to the branch target buffer. `src_addr` is the address of the
|
||||
// last halfword of a taken backward branch. The frontend redirects fetch
|
||||
// such that `src_addr` appears to be sequentially followed by `target`.
|
||||
input wire btb_set,
|
||||
input wire [W_ADDR-1:0] btb_set_src_addr,
|
||||
input wire btb_set_src_size,
|
||||
input wire [W_ADDR-1:0] btb_set_target_addr,
|
||||
input wire btb_clear,
|
||||
output wire [W_ADDR-1:0] btb_target_addr_out,
|
||||
|
||||
// Interface to Decode
|
||||
output reg [31:0] cir, // Current instruction register; pre-expanded to 32-bit
|
||||
output wire [31:0] cir_raw, // Unexpanded instruction data
|
||||
output reg [1:0] cir_vld, // number of valid halfwords in CIR
|
||||
input wire [1:0] cir_use, // number of halfwords D intends to consume
|
||||
// *may* be a function of hready
|
||||
output wire [1:0] cir_err, // Bus error on upper/lower halfword of CIR.
|
||||
output wire [1:0] cir_predbranch, // Set for last halfword of a predicted-taken branch
|
||||
output wire cir_break_any, // Set for exact match of a breakpoint address on CIR LSB
|
||||
output wire cir_break_d_mode, // As above but specifically break to debug mode
|
||||
output reg cir_is_32bit, // Can't be decoded from CIR due to pre-expansion
|
||||
output reg cir_invalid_16bit, // Expanded an invalid 32-bit instruction
|
||||
output reg cir_is_uop, // Current instruction is part of a micro-op sequence
|
||||
output reg cir_uop_nonfinal, // ...and there are more to follow in this instruction
|
||||
output reg cir_uop_no_pc_update, // Suppress PC increment or jump (note the jump in cm.popret is not the final uop!)
|
||||
output reg cir_uop_atomic, // Prevent IRQ entry, so intermediate states are not observed
|
||||
input wire uop_stall,
|
||||
input wire uop_clear,
|
||||
|
||||
// "flush_behind": do not flush the oldest instruction when accepting a
|
||||
// jump request (but still flush younger instructions). Sometimes a
|
||||
// stalled instruction may assert a jump request, because e.g. the stall
|
||||
// is dependent on a bus stall signal so can't gate the request.
|
||||
input wire cir_flush_behind,
|
||||
// Required for regnum predecode when Zilsd is enabled:
|
||||
input wire df_lspair_phase_next,
|
||||
|
||||
// Signal to power controller that power down is safe. (When going to
|
||||
// sleep, first the pipeline is stalled, and then the power controller
|
||||
// waits for the frontend to naturally come to a halt before releasing
|
||||
// its power request. This avoids manually halting the frontend.)
|
||||
output wire pwrdown_ok,
|
||||
// Signal to delay the first instruction fetch following reset, because
|
||||
// powerup has not yet been negotiated.
|
||||
input wire delay_first_fetch,
|
||||
|
||||
// Provide the rs1/rs2 register numbers which will be in CIR next cycle.
|
||||
// Coarse: valid if this instruction has a nonzero register operand.
|
||||
// (Suitable for regfile read)
|
||||
output reg [4:0] predecode_rs1_coarse,
|
||||
output reg [4:0] predecode_rs2_coarse,
|
||||
// Fine: like coarse, but accurate zeroing when the operand is implicit.
|
||||
// (Suitable for bypass. Still not precise enough for stall logic.)
|
||||
output reg [4:0] predecode_rs1_fine,
|
||||
output reg [4:0] predecode_rs2_fine,
|
||||
|
||||
// Debugger instruction injection: instruction fetch is suppressed when in
|
||||
// debug halt state, and the DM can then inject instructions into the last
|
||||
// entry of the prefetch queue using the vld/rdy handshake.
|
||||
input wire debug_mode,
|
||||
input wire [W_DATA-1:0] dbg_instr_data,
|
||||
input wire dbg_instr_data_vld,
|
||||
output wire dbg_instr_data_rdy,
|
||||
|
||||
// PMP query->kill interface for X permission checks
|
||||
output wire [W_ADDR-1:0] pmp_i_addr,
|
||||
output wire pmp_i_m_mode,
|
||||
input wire pmp_i_kill,
|
||||
|
||||
// Trigger unit query->break interface for breakpoints
|
||||
output wire [W_ADDR-1:0] trigger_addr,
|
||||
output wire trigger_m_mode,
|
||||
input wire [1:0] trigger_break_any,
|
||||
input wire [1:0] trigger_break_d_mode
|
||||
|
||||
);
|
||||
|
||||
`include "rv_opcodes.vh"
|
||||
|
||||
localparam W_BUNDLE = 16;
|
||||
// This is the minimum for full throughput (enough to avoid dropping data when
|
||||
// decode stalls) and there is no significant advantage to going larger.
|
||||
localparam FIFO_DEPTH = 2;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Fetch queue
|
||||
|
||||
wire jump_now = jump_target_vld && jump_target_rdy;
|
||||
reg [1:0] mem_data_hwvld;
|
||||
|
||||
// PMP X faults are checked in parallel with the fetch (fine if executable
|
||||
// memory is read-idempotent) and failures are promoted to bus errors:
|
||||
wire pmp_kill_fetch_dph;
|
||||
wire mem_or_pmp_err = mem_data_err || pmp_kill_fetch_dph;
|
||||
|
||||
// Similarly, breakpoint matches are checked during fetch data phase. These
|
||||
// are called mem_xxx because they are the breakpoint metadata for the data
|
||||
// coming back from memory in this dphase.
|
||||
wire [1:0] mem_break_any;
|
||||
wire [1:0] mem_break_d_mode;
|
||||
|
||||
// Mark data as containing a predicted-taken branch instruction so that
|
||||
// mispredicts can be recovered -- need to track both halfwords so that we
|
||||
// can mark the entire instruction, and nothing but the instruction:
|
||||
reg [1:0] mem_data_predbranch;
|
||||
|
||||
// Bus errors (and other metadata) travel alongside data. They cause an
|
||||
// exception if the core decodes the instruction, but until then can be
|
||||
// flushed harmlessly.
|
||||
|
||||
reg [W_DATA-1:0] fifo_mem [0:FIFO_DEPTH];
|
||||
reg fifo_err [0:FIFO_DEPTH];
|
||||
reg [1:0] fifo_break_any [0:FIFO_DEPTH];
|
||||
reg [1:0] fifo_break_d_mode [0:FIFO_DEPTH];
|
||||
reg [1:0] fifo_predbranch [0:FIFO_DEPTH];
|
||||
reg [1:0] fifo_valid_hw [0:FIFO_DEPTH];
|
||||
reg fifo_valid [0:FIFO_DEPTH];
|
||||
reg fifo_valid_m1 [0:FIFO_DEPTH];
|
||||
|
||||
wire [W_DATA-1:0] fifo_rdata = fifo_mem[0];
|
||||
wire fifo_full = fifo_valid[FIFO_DEPTH - 1];
|
||||
wire fifo_empty = !fifo_valid[0];
|
||||
wire fifo_almost_full = fifo_valid[FIFO_DEPTH - 2];
|
||||
|
||||
wire fifo_push;
|
||||
wire fifo_pop;
|
||||
wire fifo_dbg_inject = DEBUG_SUPPORT && dbg_instr_data_vld && dbg_instr_data_rdy;
|
||||
|
||||
always @ (*) begin: boundary_conditions
|
||||
integer i;
|
||||
fifo_mem[FIFO_DEPTH] = mem_data;
|
||||
fifo_predbranch[FIFO_DEPTH] = 2'b00;
|
||||
fifo_err[FIFO_DEPTH] = 1'b0;
|
||||
fifo_break_any[FIFO_DEPTH] = 2'b00;
|
||||
fifo_break_d_mode[FIFO_DEPTH] = 2'b00;
|
||||
for (i = 0; i <= FIFO_DEPTH; i = i + 1) begin
|
||||
fifo_valid[i] = |EXTENSION_C ? |fifo_valid_hw[i] : fifo_valid_hw[i][0];
|
||||
// valid-to-right condition: i == 0 || fifo_valid[i - 1], but without
|
||||
// using negative array bound (seems broken in Yosys?) or OOB in the
|
||||
// short circuit case (gives lint although result is well-defined)
|
||||
if (i == 0) begin
|
||||
fifo_valid_m1[i] = 1'b1;
|
||||
end else begin
|
||||
fifo_valid_m1[i] = fifo_valid[i - 1];
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin: fifo_update
|
||||
integer i;
|
||||
if (!rst_n) begin
|
||||
for (i = 0; i < FIFO_DEPTH; i = i + 1) begin
|
||||
fifo_valid_hw[i] <= 2'b00;
|
||||
fifo_mem[i] <= 32'd0;
|
||||
fifo_err[i] <= 1'b0;
|
||||
fifo_break_any[i] <= 2'b00;
|
||||
fifo_break_d_mode[i] <= 2'b00;
|
||||
fifo_predbranch[i] <= 2'b00;
|
||||
end
|
||||
// This exists only for loop boundary conditions, but is tied off in
|
||||
// this synchronous process to work around a Verilator scheduling
|
||||
// issue (see issue #21)
|
||||
fifo_valid_hw[FIFO_DEPTH] <= 2'b00;
|
||||
end else begin
|
||||
for (i = 0; i < FIFO_DEPTH; i = i + 1) begin
|
||||
if (fifo_pop || (fifo_push && !fifo_valid[i])) begin
|
||||
fifo_mem[i] <= fifo_valid[i + 1] ? fifo_mem[i + 1] : mem_data;
|
||||
fifo_err[i] <= fifo_valid[i + 1] ? fifo_err[i + 1] : mem_or_pmp_err;
|
||||
fifo_break_any[i] <= fifo_valid[i + 1] ? fifo_break_any[i + 1] : mem_break_any;
|
||||
fifo_break_d_mode[i] <= fifo_valid[i + 1] ? fifo_break_d_mode[i + 1] : mem_break_d_mode;
|
||||
fifo_predbranch[i] <= fifo_valid[i + 1] ? fifo_predbranch[i + 1] : mem_data_predbranch;
|
||||
end
|
||||
fifo_valid_hw[i] <=
|
||||
jump_now ? 2'h0 :
|
||||
fifo_valid[i + 1] && fifo_pop ? fifo_valid_hw[i + 1] :
|
||||
fifo_valid[i] && fifo_pop ? mem_data_hwvld & {2{fifo_push}} :
|
||||
fifo_valid[i] ? fifo_valid_hw[i] :
|
||||
fifo_push && !fifo_pop && fifo_valid_m1[i] ? mem_data_hwvld : 2'h0;
|
||||
end
|
||||
// Allow DM to inject instructions directly into the lowest-numbered
|
||||
// queue entry. This mux should not extend critical path since it is
|
||||
// balanced with the instruction-assembly muxes on the queue bypass
|
||||
// path. Note that flush takes precedence over debug injection
|
||||
// (and the debug module design must account for this)
|
||||
if (fifo_dbg_inject) begin
|
||||
fifo_mem[0] <= dbg_instr_data;
|
||||
fifo_err[0] <= 1'b0;
|
||||
fifo_predbranch[0] <= 2'b00;
|
||||
fifo_break_any[0] <= 2'b00;
|
||||
fifo_break_d_mode[0] <= 2'b00;
|
||||
fifo_valid_hw[0] <= jump_now ? 2'b00 : 2'b11;
|
||||
end
|
||||
fifo_valid_hw[FIFO_DEPTH] <= 2'b00;
|
||||
end
|
||||
end
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
always @ (posedge clk) if (rst_n) begin
|
||||
// FIFO validity must be compact, so we can always consume from the end
|
||||
if (!fifo_valid[0]) begin
|
||||
assert(!fifo_valid[1]);
|
||||
end
|
||||
end
|
||||
`endif
|
||||
|
||||
assign pwrdown_ok = (fifo_full && !jump_target_vld) || debug_mode;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Branch target buffer
|
||||
|
||||
wire [W_ADDR-1:0] btb_src_addr;
|
||||
wire btb_src_size;
|
||||
wire [W_ADDR-1:0] btb_target_addr;
|
||||
wire btb_valid;
|
||||
|
||||
generate
|
||||
if (BRANCH_PREDICTOR) begin: have_btb
|
||||
reg [W_ADDR-1:0] btb_src_addr_r;
|
||||
reg btb_src_size_r;
|
||||
reg [W_ADDR-1:0] btb_target_addr_r;
|
||||
reg btb_valid_r;
|
||||
assign btb_src_addr = btb_src_addr_r;
|
||||
assign btb_src_size = btb_src_size_r;
|
||||
assign btb_target_addr = btb_target_addr_r;
|
||||
assign btb_valid = btb_valid_r;
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
btb_src_addr_r <= {W_ADDR{1'b0}};
|
||||
btb_src_size_r <= 1'b0;
|
||||
btb_target_addr_r <= {W_ADDR{1'b0}};
|
||||
btb_valid_r <= 1'b0;
|
||||
end else if (btb_clear) begin
|
||||
// Clear takes precedences over set. E.g. if a taken branch is in
|
||||
// stage 2 and an exception is in stage 3, we must clear the BTB.
|
||||
btb_valid_r <= 1'b0;
|
||||
end else if (btb_set) begin
|
||||
btb_src_addr_r <= btb_set_src_addr;
|
||||
btb_src_size_r <= btb_set_src_size;
|
||||
btb_target_addr_r <= btb_set_target_addr;
|
||||
btb_valid_r <= 1'b1;
|
||||
end
|
||||
end
|
||||
end else begin: no_btb
|
||||
assign btb_src_addr = {W_ADDR{1'b0}};
|
||||
assign btb_src_size = 1'b0;
|
||||
assign btb_target_addr = {W_ADDR{1'b0}};
|
||||
assign btb_valid = 1'b0;
|
||||
end
|
||||
endgenerate
|
||||
|
||||
// Decode uses the target address to set the PC to the correct branch target
|
||||
// value following a predicted-taken branch (as normally it would update PC
|
||||
// by following an X jump request, and in this case there is none).
|
||||
//
|
||||
// Note this assumes the BTB target has not changed by the time the predicted
|
||||
// branch arrives at decode! This is always true because the only way for the
|
||||
// target address to change is when an older branch is taken, which would
|
||||
// flush the younger predicted-taken branch before it reaches decode.
|
||||
|
||||
assign btb_target_addr_out = btb_target_addr;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Fetch request generation
|
||||
|
||||
// Fetch addr runs ahead of the PC, in word increments.
|
||||
reg [W_ADDR-1:0] fetch_addr;
|
||||
reg fetch_priv;
|
||||
reg btb_prev_start_of_overhanging;
|
||||
reg [1:0] mem_aph_hwvld;
|
||||
reg mem_addr_hold;
|
||||
|
||||
wire btb_match_word = |BRANCH_PREDICTOR && btb_valid && (
|
||||
fetch_addr[W_ADDR-1:2] == btb_src_addr[W_ADDR-1:2]
|
||||
);
|
||||
|
||||
// Catch case where predicted-taken branch instruction extends into next word:
|
||||
wire btb_src_overhanging = btb_src_size && btb_src_addr[1];
|
||||
|
||||
// Suppress case where we have jumped immediately after a word-aligned halfword-sized
|
||||
// branch, and the jump target went into fetch_addr due to an address-phase hold:
|
||||
wire btb_jumped_beyond = !btb_src_size && !btb_src_addr[1] && !mem_aph_hwvld[0];
|
||||
|
||||
wire btb_match_current_addr = btb_match_word && !btb_src_overhanging && !btb_jumped_beyond;
|
||||
wire btb_match_next_addr = btb_match_word && btb_src_overhanging;
|
||||
|
||||
wire btb_match_now = btb_match_current_addr || btb_prev_start_of_overhanging;
|
||||
|
||||
// Post-increment if jump request is going straight through
|
||||
wire [W_ADDR-1:0] jump_target_post_increment =
|
||||
{jump_target[W_ADDR-1:2], 2'b00} +
|
||||
{{W_ADDR-3{1'b0}}, mem_addr_rdy && !mem_addr_hold, 2'b00};
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
fetch_addr <= RESET_VECTOR;
|
||||
// M-mode at reset:
|
||||
fetch_priv <= 1'b1;
|
||||
btb_prev_start_of_overhanging <= 1'b0;
|
||||
end else begin
|
||||
if (jump_now) begin
|
||||
fetch_addr <= jump_target_post_increment;
|
||||
fetch_priv <= jump_priv || !U_MODE;
|
||||
btb_prev_start_of_overhanging <= 1'b0;
|
||||
end else if (mem_addr_vld && mem_addr_rdy) begin
|
||||
if (btb_match_now && |BRANCH_PREDICTOR) begin
|
||||
fetch_addr <= {btb_target_addr[W_ADDR-1:2], 2'b00};
|
||||
end else begin
|
||||
fetch_addr <= fetch_addr + 32'd4;
|
||||
end
|
||||
btb_prev_start_of_overhanging <= btb_match_next_addr;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// Combinatorially generate the address-phase request
|
||||
|
||||
reg reset_holdoff;
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
reset_holdoff <= 1'b1;
|
||||
end else begin
|
||||
reset_holdoff <= (|EXTENSION_XH3POWER && delay_first_fetch) ? reset_holdoff : 1'b0;
|
||||
// This should be impossible, but assert to be sure, because it *will*
|
||||
// change the fetch address (and we shouldn't check it in hardware if
|
||||
// we can prove it doesn't happen)
|
||||
end
|
||||
end
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
always @ (posedge clk) if (rst_n) begin
|
||||
assert(!(jump_target_vld && reset_holdoff));
|
||||
end
|
||||
`endif
|
||||
|
||||
reg [W_ADDR-1:0] mem_addr_r;
|
||||
reg mem_priv_r;
|
||||
reg mem_addr_vld_r;
|
||||
|
||||
// Downstream accesses are always word-sized word-aligned.
|
||||
assign mem_addr = mem_addr_r;
|
||||
assign mem_priv = mem_priv_r;
|
||||
assign mem_addr_vld = mem_addr_vld_r && !reset_holdoff;
|
||||
assign mem_size = 1'b1;
|
||||
|
||||
wire fetch_stall;
|
||||
|
||||
always @ (*) begin
|
||||
mem_addr_r = fetch_addr;
|
||||
mem_priv_r = fetch_priv;
|
||||
mem_addr_vld_r = 1'b1;
|
||||
case (1'b1)
|
||||
mem_addr_hold : begin mem_addr_r = fetch_addr; end
|
||||
jump_target_vld || reset_holdoff : begin
|
||||
mem_addr_r = {jump_target[W_ADDR-1:2], 2'b00};
|
||||
mem_priv_r = jump_priv || !U_MODE;
|
||||
end
|
||||
DEBUG_SUPPORT && debug_mode : begin mem_addr_vld_r = 1'b0; end
|
||||
!fetch_stall : begin mem_addr_r = fetch_addr; end
|
||||
default : begin mem_addr_vld_r = 1'b0; end
|
||||
endcase
|
||||
end
|
||||
|
||||
assign jump_target_rdy = !mem_addr_hold;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Bus Pipeline Tracking
|
||||
|
||||
// Keep track of some useful state of the memory interface
|
||||
|
||||
reg [1:0] pending_fetches;
|
||||
reg [1:0] ctr_flush_pending;
|
||||
|
||||
wire [1:0] pending_fetches_next = pending_fetches + (mem_addr_vld && !mem_addr_hold) - mem_data_vld;
|
||||
|
||||
// Using the non-registered version of pending_fetches would improve FIFO
|
||||
// utilisation, but create a combinatorial path from hready to address phase!
|
||||
// This means at least a 2-word FIFO is required for full fetch throughput.
|
||||
assign fetch_stall = fifo_full
|
||||
|| fifo_almost_full && |pending_fetches
|
||||
|| pending_fetches > 2'h1;
|
||||
|
||||
// Debugger only injects instructions when the frontend is at rest and empty.
|
||||
assign dbg_instr_data_rdy = DEBUG_SUPPORT && !fifo_valid[0] && ~|ctr_flush_pending;
|
||||
|
||||
wire cir_room_for_fetch;
|
||||
// If fetch data is forwarded past the FIFO, ensure it is not also written to it.
|
||||
assign fifo_push = mem_data_vld && ~|ctr_flush_pending && !(cir_room_for_fetch && fifo_empty)
|
||||
&& !(DEBUG_SUPPORT && debug_mode);
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
mem_addr_hold <= 1'b0;
|
||||
pending_fetches <= 2'h0;
|
||||
ctr_flush_pending <= 2'h0;
|
||||
end else begin
|
||||
mem_addr_hold <= mem_addr_vld && !mem_addr_rdy;
|
||||
pending_fetches <= pending_fetches_next;
|
||||
if (jump_now) begin
|
||||
ctr_flush_pending <= pending_fetches - mem_data_vld;
|
||||
end else if (|ctr_flush_pending && mem_data_vld) begin
|
||||
ctr_flush_pending <= ctr_flush_pending - 1'b1;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
always @ (posedge clk) if (rst_n) begin
|
||||
assert(ctr_flush_pending <= pending_fetches);
|
||||
assert(pending_fetches < 2'd3);
|
||||
assert(!(mem_data_vld && !pending_fetches));
|
||||
end
|
||||
`endif
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
mem_data_hwvld <= 2'b11;
|
||||
mem_aph_hwvld <= 2'b11;
|
||||
mem_data_predbranch <= 2'b00;
|
||||
end else begin
|
||||
if (jump_now) begin
|
||||
if (|EXTENSION_C) begin
|
||||
if (mem_addr_rdy) begin
|
||||
mem_aph_hwvld <= 2'b11;
|
||||
mem_data_hwvld <= {1'b1, !jump_target[1]};
|
||||
end else begin
|
||||
mem_aph_hwvld <= {1'b1, !jump_target[1]};
|
||||
end
|
||||
end
|
||||
mem_data_predbranch <= 2'b00;
|
||||
end else if (mem_addr_vld && mem_addr_rdy) begin
|
||||
if (|EXTENSION_C) begin
|
||||
// If a predicted-taken branch instruction only spans the first
|
||||
// half of a word, need to flag the second half as invalid.
|
||||
mem_data_hwvld <= mem_aph_hwvld & {
|
||||
!(|BRANCH_PREDICTOR && btb_match_now && (btb_src_addr[1] == btb_src_size)),
|
||||
1'b1
|
||||
};
|
||||
// Also need to take the alignment of the destination into account.
|
||||
mem_aph_hwvld <= {
|
||||
1'b1,
|
||||
!(|BRANCH_PREDICTOR && btb_match_now && btb_target_addr[1])
|
||||
};
|
||||
end
|
||||
mem_data_predbranch <=
|
||||
|BRANCH_PREDICTOR && btb_match_word ? (
|
||||
btb_src_addr[1] ? 2'b10 :
|
||||
btb_src_size ? 2'b11 : 2'b01
|
||||
) :
|
||||
|BRANCH_PREDICTOR && btb_prev_start_of_overhanging ? (
|
||||
2'b01
|
||||
) : 2'b00;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// PMP and trigger unit interfacing: query -> kill/break
|
||||
|
||||
wire [W_ADDR-1:0] pmp_trigger_check_dph_addr;
|
||||
wire pmp_trigger_check_dph_m_mode;
|
||||
|
||||
// Register the fetch address into stage F so that the PMP can check it in
|
||||
// parallel with the bus data phase. Feels wasteful to have a separate
|
||||
// register, but using the fetch_addr counter is fraught due to the way that
|
||||
// new addresses go into it or past it (depending on aphase hold).
|
||||
|
||||
generate
|
||||
if (PMP_REGIONS > 0 || DEBUG_SUPPORT != 0) begin: have_check_reg
|
||||
|
||||
reg [W_ADDR-1:0] check_addr_dph;
|
||||
reg check_m_mode_dph;
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
check_addr_dph <= {W_ADDR{1'b0}};
|
||||
check_m_mode_dph <= 1'b0;
|
||||
end else if (mem_addr_vld && mem_addr_rdy) begin
|
||||
check_addr_dph <= mem_addr;
|
||||
check_m_mode_dph <= mem_priv;
|
||||
end
|
||||
end
|
||||
|
||||
assign pmp_trigger_check_dph_addr = check_addr_dph;
|
||||
assign pmp_trigger_check_dph_m_mode = check_m_mode_dph;
|
||||
|
||||
end else begin: no_check_reg
|
||||
|
||||
assign pmp_trigger_check_dph_addr = {W_ADDR{1'b0}};
|
||||
assign pmp_trigger_check_dph_m_mode = 1'b0;
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
generate
|
||||
if (PMP_REGIONS == 0) begin: no_pmp
|
||||
|
||||
assign pmp_i_addr = {W_ADDR{1'b0}};
|
||||
assign pmp_i_m_mode = 1'b0;
|
||||
assign pmp_kill_fetch_dph = 1'b0;
|
||||
|
||||
end else begin: have_pmp
|
||||
|
||||
assign pmp_i_addr = pmp_trigger_check_dph_addr;
|
||||
assign pmp_i_m_mode = pmp_trigger_check_dph_m_mode;
|
||||
assign pmp_kill_fetch_dph = pmp_i_kill && !debug_mode;
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
generate
|
||||
if (DEBUG_SUPPORT == 0) begin: no_triggers
|
||||
|
||||
assign trigger_addr = {W_ADDR{1'b0}};
|
||||
assign trigger_m_mode = 1'b0;
|
||||
assign mem_break_any = 2'b00;
|
||||
assign mem_break_d_mode = 2'b00;
|
||||
|
||||
end else begin: have_triggers
|
||||
|
||||
assign trigger_addr = pmp_trigger_check_dph_addr;
|
||||
assign trigger_m_mode = pmp_trigger_check_dph_m_mode;
|
||||
assign mem_break_any = trigger_break_any & {|EXTENSION_C, 1'b1};
|
||||
assign mem_break_d_mode = trigger_break_d_mode & {|EXTENSION_C, 1'b1};
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Instruction buffer
|
||||
|
||||
// The instruction buffer is a 3 x ~16-bit shift register:
|
||||
//
|
||||
// * 2 x 16-bit entries form the 32-bit current instruction register (CIR)
|
||||
// which is the processor's decode window
|
||||
//
|
||||
// * 1 x 16-bit entry allows the decode window to be non-32-bit-aligned with
|
||||
// respect to the 2 x 32-bit prefetch queue entries, which are always
|
||||
// naturally aligned in memory (if fully populated).
|
||||
//
|
||||
// The third entry should be trimmed for non-RVC configurations due to
|
||||
// constant-folding on EXTENSION_C; it is unnecessary here because the
|
||||
// instructions are always 32-bit-aligned.
|
||||
|
||||
// The entries ("slots") are slightly larger than 16 bits because they also
|
||||
// contain metadata like bus errors:
|
||||
localparam W_SLOT = 4 + W_BUNDLE;
|
||||
localparam SLOT_BREAK_ANY_BIT = 3 + W_BUNDLE;
|
||||
localparam SLOT_BREAK_D_MODE_BIT = 2 + W_BUNDLE;
|
||||
localparam SLOT_ERR_BIT = 1 + W_BUNDLE;
|
||||
localparam SLOT_PREDBRANCH_BIT = 0 + W_BUNDLE;
|
||||
|
||||
reg [3*W_SLOT-1:0] buf_contents;
|
||||
reg [1:0] buf_level;
|
||||
|
||||
wire fetch_data_vld = !fifo_empty || (mem_data_vld && ~|ctr_flush_pending && !debug_mode);
|
||||
|
||||
wire [W_DATA-1:0] fetch_data = fifo_empty ? mem_data : fifo_rdata;
|
||||
wire [1:0] fetch_data_hwvld = fifo_empty ? mem_data_hwvld : fifo_valid_hw[0];
|
||||
wire fetch_bus_err = fifo_empty ? mem_or_pmp_err : fifo_err[0];
|
||||
wire [1:0] fetch_break_any = fifo_empty ? mem_break_any : fifo_break_any[0];
|
||||
wire [1:0] fetch_break_d_mode = fifo_empty ? mem_break_d_mode : fifo_break_d_mode[0];
|
||||
wire [1:0] fetch_predbranch = fifo_empty ? mem_data_predbranch : fifo_predbranch[0];
|
||||
|
||||
wire [W_SLOT-1:0] fetch_contents_hw1 = {
|
||||
fetch_break_any[1],
|
||||
fetch_break_d_mode[1],
|
||||
fetch_bus_err,
|
||||
fetch_predbranch[1],
|
||||
fetch_data[W_BUNDLE +: W_BUNDLE]
|
||||
};
|
||||
|
||||
wire [W_SLOT-1:0] fetch_contents_hw0 = {
|
||||
fetch_break_any[0],
|
||||
fetch_break_d_mode[0],
|
||||
fetch_bus_err,
|
||||
fetch_predbranch[0],
|
||||
fetch_data[0 +: W_BUNDLE]
|
||||
};
|
||||
|
||||
wire [2*W_SLOT-1:0] fetch_contents_aligned = {
|
||||
fetch_contents_hw1,
|
||||
fetch_data_hwvld[0] || ~|EXTENSION_C ? fetch_contents_hw0 : fetch_contents_hw1
|
||||
};
|
||||
|
||||
// Shift not-yet-used contents down to backfill D's consumption. We don't care
|
||||
// about anything which is invalid or will be overlaid with fresh data, so
|
||||
// choose these values in a way that minimises muxes.
|
||||
wire [3*W_SLOT-1:0] buf_shifted =
|
||||
cir_use[1] ? {buf_contents[W_SLOT +: 2 * W_SLOT], buf_contents[2 * W_SLOT +: W_SLOT]} :
|
||||
cir_use[0] && EXTENSION_C ? {buf_contents[2 * W_SLOT +: W_SLOT], buf_contents[W_SLOT +: 2 * W_SLOT]} :
|
||||
buf_contents;
|
||||
|
||||
wire [1:0] level_next_no_fetch = buf_level - cir_use;
|
||||
|
||||
// Overlay fresh fetch data onto the shifted/recycled buffer contents. Again,
|
||||
// if something won't be looked at, generate the cheapest possible garbage.
|
||||
assign cir_room_for_fetch = level_next_no_fetch <= (|EXTENSION_C && ~&fetch_data_hwvld ? 2'h2 : 2'h1);
|
||||
assign fifo_pop = cir_room_for_fetch && !fifo_empty;
|
||||
|
||||
wire [3*W_SLOT-1:0] buf_shifted_plus_fetch =
|
||||
!cir_room_for_fetch ? buf_shifted :
|
||||
level_next_no_fetch[1] && |EXTENSION_C ? {fetch_contents_aligned[0 +: W_SLOT], buf_shifted[0 +: 2 * W_SLOT]} :
|
||||
level_next_no_fetch[0] && |EXTENSION_C ? {fetch_contents_aligned, buf_shifted[0 +: W_SLOT]} :
|
||||
{buf_shifted[2 * W_SLOT +: W_SLOT], fetch_contents_aligned};
|
||||
|
||||
wire [1:0] fetch_fill_amount = cir_room_for_fetch && fetch_data_vld ? (
|
||||
&fetch_data_hwvld || ~|EXTENSION_C ? 2'h2 : 2'h1
|
||||
) : 2'h0;
|
||||
|
||||
wire [1:0] buf_level_next = {1'b1, |EXTENSION_C} & (
|
||||
jump_now && cir_flush_behind ? (cir_is_32bit ? 2'h2 : 2'h1) :
|
||||
jump_now ? 2'h0 : level_next_no_fetch + fetch_fill_amount
|
||||
);
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
buf_level <= 2'h0;
|
||||
cir_vld <= 2'h0;
|
||||
// Mysterious reset value ensures address buses are zero in reset
|
||||
// (see definition of d_addr_offs in hazard3_decode)
|
||||
buf_contents <= {{3 * W_SLOT - 2{1'b0}}, 2'b11};
|
||||
end else begin
|
||||
buf_level <= buf_level_next;
|
||||
cir_vld <= buf_level_next & ~(buf_level_next >> 1'b1);
|
||||
buf_contents <= buf_shifted_plus_fetch;
|
||||
end
|
||||
end
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
reg [1:0] prop_past_buf_level; // Workaround for weird non-constant $past reset issue
|
||||
always @ (posedge clk) begin
|
||||
if (!rst_n) begin
|
||||
prop_past_buf_level <= 2'h0;
|
||||
end else begin
|
||||
prop_past_buf_level <= buf_level;
|
||||
|
||||
assert(cir_vld <= 2);
|
||||
assert(cir_use <= cir_vld);
|
||||
if (!jump_now) assert(buf_level_next >= level_next_no_fetch);
|
||||
// We fetch 32 bits per cycle, max. If this happens it's due to negative overflow.
|
||||
if (prop_past_buf_level == 2'h0)
|
||||
assert(buf_level != 2'h3);
|
||||
end
|
||||
end
|
||||
`endif
|
||||
|
||||
assign cir_err = {
|
||||
buf_contents[1 * W_SLOT + SLOT_ERR_BIT],
|
||||
buf_contents[0 * W_SLOT + SLOT_ERR_BIT]
|
||||
};
|
||||
|
||||
assign cir_predbranch = {
|
||||
buf_contents[1 * W_SLOT + SLOT_PREDBRANCH_BIT],
|
||||
buf_contents[0 * W_SLOT + SLOT_PREDBRANCH_BIT]
|
||||
};
|
||||
|
||||
assign cir_break_any = buf_contents[0 * W_SLOT + SLOT_BREAK_ANY_BIT] && |cir_vld;
|
||||
|
||||
assign cir_break_d_mode = buf_contents[0 * W_SLOT + SLOT_BREAK_D_MODE_BIT] && |cir_vld;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Register number predecode
|
||||
|
||||
wire [31:0] next_instr = {
|
||||
buf_shifted_plus_fetch[1 * W_SLOT +: W_BUNDLE],
|
||||
buf_shifted_plus_fetch[0 * W_SLOT +: W_BUNDLE]
|
||||
};
|
||||
|
||||
wire next_instr_is_32bit = next_instr[1:0] == 2'b11 || ~|EXTENSION_C;
|
||||
|
||||
wire [3:0] decomp_uop_step;
|
||||
wire [3:0] uop_ctr = decomp_uop_step & {4{|EXTENSION_ZCMP}};
|
||||
|
||||
wire [4:0] zcmp_pushpop_rs2 =
|
||||
uop_ctr == 4'h0 ? 5'd01 : // ra
|
||||
uop_ctr == 4'h1 ? 5'd08 : // s0
|
||||
uop_ctr == 4'h2 ? 5'd09 : // s1
|
||||
5'd15 + {1'b0, uop_ctr} ; // s2-s11
|
||||
|
||||
wire [4:0] zcmp_pushpop_rs1 =
|
||||
uop_ctr < 4'hd ? 5'd02 : // sp (addr base reg)
|
||||
uop_ctr == 4'hd ? 5'd00 : // zero (clear a0)
|
||||
uop_ctr == 4'he ? 5'd01 : // ra (ret)
|
||||
5'd02 ; // sp (stack adj)
|
||||
|
||||
wire [4:0] zcmp_sa01_r1s = {|next_instr[9:8], ~|next_instr[9:8], next_instr[9:7]};
|
||||
wire [4:0] zcmp_sa01_r2s = {|next_instr[4:3], ~|next_instr[4:3], next_instr[4:2]};
|
||||
|
||||
wire [4:0] zcmp_mvsa01_rs1 = {4'h5, uop_ctr[0]};
|
||||
wire [4:0] zcmp_mva01s_rs1 = uop_ctr[0] ? zcmp_sa01_r2s : zcmp_sa01_r1s;
|
||||
|
||||
// "coarse" because the mapping of pair (x0, x1) -> (x0, x0) is not yet applied
|
||||
wire [4:0] zilsd_rs2_coarse = { next_instr[24:21], df_lspair_phase_next ^ ~next_instr[15]};
|
||||
wire [4:0] zclsd_sd_rs2_coarse = {2'b01, next_instr[4:3], df_lspair_phase_next ^ ~next_instr[7] };
|
||||
wire [4:0] zclsd_sdsp_rs2_coarse = { next_instr[6:3], df_lspair_phase_next ^ 1'b1 };
|
||||
|
||||
always @ (*) begin
|
||||
|
||||
casez ({next_instr_is_32bit, |EXTENSION_ZCMP, next_instr[15:0]})
|
||||
{1'b1, 1'bz, 16'bzzzzzzzzzzzzzzzz}: predecode_rs1_coarse = next_instr[19:15]; // 32-bit R, S, B formats
|
||||
{1'b0, 1'bz, 16'b00zzzzzzzzzzzz00}: predecode_rs1_coarse = 5'd2; // c.addi4spn + don't care
|
||||
{1'b0, 1'bz, 16'b0zzzzzzzzzzzzz01}: predecode_rs1_coarse = next_instr[11:7]; // c.addi, c.addi16sp + don't care (jal, li)
|
||||
{1'b0, 1'bz, 16'bz1zzzzzzzzzzzz10}: predecode_rs1_coarse = 5'd2; // c.lwsp, c.swsp, c.ldsp, c.sdsp
|
||||
{1'b0, 1'bz, 16'bz00zzzzzzzzzzz10}: predecode_rs1_coarse = next_instr[11:7]; // c.slli, c.mv, c.add
|
||||
{1'b0, 1'b1, 16'b1011zzzzzzzzzz10}: predecode_rs1_coarse = zcmp_pushpop_rs1; // cm.push, cm.pop*
|
||||
{1'b0, 1'b1, 16'b1010zzzzz0zzzz10}: predecode_rs1_coarse = zcmp_mvsa01_rs1; // cm.mvsa01
|
||||
{1'b0, 1'b1, 16'b1010zzzzz1zzzz10}: predecode_rs1_coarse = zcmp_mva01s_rs1; // cm.mva01s
|
||||
default: predecode_rs1_coarse = {2'b01, next_instr[9:7]};
|
||||
endcase
|
||||
|
||||
casez ({next_instr_is_32bit, |EXTENSION_ZCMP, |EXTENSION_ZILSD, |EXTENSION_ZCLSD, next_instr[15:0]})
|
||||
{1'b1, 1'bz, 1'b1, 1'bz, 16'bzz11zzzzz0z0zzzz}: predecode_rs2_coarse = zilsd_rs2_coarse; // ld, sd (Zilsd)
|
||||
{1'b1, 1'bz, 1'b0, 1'bz, 16'bzz11zzzzz0z0zzzz}: predecode_rs2_coarse = next_instr[24:20]; // ld, sd (no Zilsd)
|
||||
|
||||
{1'b1, 1'bz, 1'bz, 1'bz, 16'bzz0zzzzzzzzzzzzz}: predecode_rs2_coarse = next_instr[24:20]; // (cover remaining 32-bit
|
||||
{1'b1, 1'bz, 1'bz, 1'bz, 16'bzz10zzzzzzzzzzzz}: predecode_rs2_coarse = next_instr[24:20]; // patterns, without overlap)
|
||||
{1'b1, 1'bz, 1'bz, 1'bz, 16'bzz11zzzzz1zzzzzz}: predecode_rs2_coarse = next_instr[24:20];
|
||||
{1'b1, 1'bz, 1'bz, 1'bz, 16'bzz11zzzzz0z1zzzz}: predecode_rs2_coarse = next_instr[24:20];
|
||||
|
||||
{1'b0, 1'bz, 1'b1, 1'b1, 16'bzz1zzzzzzzzzzz00}: predecode_rs2_coarse = zclsd_sd_rs2_coarse;
|
||||
{1'b0, 1'bz, 1'bz, 1'bz, 16'bzz0zzzzzzzzzzz10}: predecode_rs2_coarse = next_instr[6:2]; // c.add, c.swsp
|
||||
{1'b0, 1'b1, 1'bz, 1'bz, 16'bz01zzzzzzzzzzz10}: predecode_rs2_coarse = zcmp_pushpop_rs2; // cm.push
|
||||
{1'b0, 1'bz, 1'b1, 1'b1, 16'bz11zzzzzzzzzzz10}: predecode_rs2_coarse = zclsd_sdsp_rs2_coarse;
|
||||
default: predecode_rs2_coarse = {2'b01, next_instr[4:2]};
|
||||
endcase
|
||||
|
||||
// The "fine" predecode targets those instructions which either:
|
||||
// - Have an implicit zero-register operand in their expanded form (e.g. c.beqz)
|
||||
// - Do not have a register operand on that port, but rely on the port being 0
|
||||
// We don't care about instructions which ignore the reg ports, e.g. ebreak
|
||||
|
||||
casez ({|EXTENSION_C, next_instr})
|
||||
// -> addi rd, x0, imm:
|
||||
{1'b1, 16'hzzzz, `RVOPC_C_LI}: predecode_rs1_fine = 5'd0;
|
||||
{1'b1, 16'hzzzz, `RVOPC_C_MV}: begin
|
||||
if (next_instr[6:2] == 5'd0) begin
|
||||
// c.jr has rs1 as normal
|
||||
predecode_rs1_fine = predecode_rs1_coarse;
|
||||
end else begin
|
||||
// -> add rd, x0, rs2:
|
||||
predecode_rs1_fine = 5'd0;
|
||||
end
|
||||
end
|
||||
default: predecode_rs1_fine = predecode_rs1_coarse;
|
||||
endcase
|
||||
|
||||
casez ({|EXTENSION_C, |EXTENSION_ZILSD, |EXTENSION_ZCLSD, next_instr})
|
||||
{1'b1, 1'bz, 1'bz, 16'hzzzz, `RVOPC_C_BEQZ}: predecode_rs2_fine = 5'd0; // -> beq rs1, x0, label
|
||||
{1'b1, 1'bz, 1'bz, 16'hzzzz, `RVOPC_C_BNEZ}: predecode_rs2_fine = 5'd0; // -> bne rs1, x0, label
|
||||
{1'b1, 1'b1, 1'bz, `RVOPC_SD }: predecode_rs2_fine = predecode_rs2_coarse & {5{|next_instr[24:21]}};
|
||||
{1'b1, 1'b1, 1'b1, 16'hzzzz, `RVOPC_C_SDSP}: predecode_rs2_fine = predecode_rs2_coarse & {5{|next_instr[ 6: 3]}};
|
||||
default: predecode_rs2_fine = predecode_rs2_coarse;
|
||||
endcase
|
||||
|
||||
if (|EXTENSION_E) begin
|
||||
predecode_rs1_coarse[4] = 1'b0;
|
||||
predecode_rs2_coarse[4] = 1'b0;
|
||||
predecode_rs1_fine[4] = 1'b0;
|
||||
predecode_rs2_fine[4] = 1'b0;
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Instruction decompression
|
||||
|
||||
// Instructions are decompressed at the end of stage 1 (fetch data phase). On
|
||||
// ASIC, where the register file is synthesised with muxes, this puts
|
||||
// decompression somewhat in parallel with register file read, which uses
|
||||
// approximately decoded regnums.
|
||||
|
||||
generate
|
||||
if (~|EXTENSION_C) begin: no_decompress
|
||||
|
||||
// No decompression; instructions decoded directly from prefetch buffer
|
||||
always @ (*) begin
|
||||
cir = {
|
||||
buf_contents[1 * W_SLOT +: W_BUNDLE],
|
||||
buf_contents[0 * W_SLOT +: W_BUNDLE]
|
||||
} | 32'd3;
|
||||
cir_is_32bit = 1'b1;
|
||||
cir_invalid_16bit = ~&buf_contents[1:0];
|
||||
cir_is_uop = 1'b0;
|
||||
cir_uop_nonfinal = 1'b0;
|
||||
cir_uop_no_pc_update = 1'b0;
|
||||
cir_uop_atomic = 1'b0;
|
||||
end
|
||||
|
||||
assign decomp_uop_step = 4'h0;
|
||||
|
||||
end else begin: have_decompress
|
||||
|
||||
wire decomp_instr_is_32bit;
|
||||
wire [31:0] decomp_instr_out;
|
||||
wire decomp_is_uop;
|
||||
wire decomp_is_final_uop;
|
||||
wire decomp_uop_no_pc_update;
|
||||
wire decomp_uop_atomic;
|
||||
wire decomp_invalid;
|
||||
|
||||
wire first_uop = ~|decomp_uop_step;
|
||||
// Ensure the first uop goes straight through, as it is registered into CIR:
|
||||
wire uop_stall_non_first = first_uop ? ~|buf_level_next : uop_stall;
|
||||
// Ensure the uop counter stops at 0 after rolling over once:
|
||||
wire uop_stall_on_repeat = cir_is_uop && !cir_uop_nonfinal && ~|cir_use;
|
||||
|
||||
hazard3_instr_decompress #(
|
||||
`include "hazard3_config_inst.vh"
|
||||
) decomp (
|
||||
.clk (clk),
|
||||
.rst_n (rst_n),
|
||||
|
||||
.instr_in (next_instr),
|
||||
|
||||
.instr_is_32bit (decomp_instr_is_32bit),
|
||||
.instr_out (decomp_instr_out),
|
||||
|
||||
.instr_out_is_uop (decomp_is_uop),
|
||||
.instr_out_is_final_uop (decomp_is_final_uop),
|
||||
.instr_out_uop_no_pc_update (decomp_uop_no_pc_update),
|
||||
.instr_out_uop_atomic (decomp_uop_atomic),
|
||||
.instr_out_uop_stall (uop_stall_non_first || uop_stall_on_repeat),
|
||||
.instr_out_uop_clear (uop_clear),
|
||||
.df_uop_step (decomp_uop_step),
|
||||
|
||||
.invalid (decomp_invalid)
|
||||
);
|
||||
|
||||
wire cir_clken =
|
||||
~|cir_vld || (!cir_vld[1] && &buf_contents[1:0]) ||
|
||||
|cir_use || (|EXTENSION_ZCMP && cir_is_uop && !uop_stall) ||
|
||||
(|EXTENSION_ZCMP && uop_clear);
|
||||
|
||||
wire cir_is_uop_next = decomp_is_uop && |buf_level_next;
|
||||
wire cir_uop_nonfinal_next = cir_is_uop_next && !decomp_is_final_uop;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
cir <= 32'd3;
|
||||
cir_is_32bit <= 1'b0;
|
||||
cir_invalid_16bit <= 1'b0;
|
||||
cir_is_uop <= 1'b0;
|
||||
cir_uop_nonfinal <= 1'b0;
|
||||
cir_uop_no_pc_update <= 1'b0;
|
||||
cir_uop_atomic <= 1'b0;
|
||||
end else if (cir_clken) begin
|
||||
cir <= decomp_instr_out | 32'd3;
|
||||
cir_is_32bit <= decomp_instr_is_32bit;
|
||||
cir_invalid_16bit <= decomp_invalid;
|
||||
cir_is_uop <= |EXTENSION_ZCMP && !uop_clear && cir_is_uop_next;
|
||||
cir_uop_nonfinal <= |EXTENSION_ZCMP && !uop_clear && cir_uop_nonfinal_next;
|
||||
cir_uop_no_pc_update <= |EXTENSION_ZCMP && !uop_clear && decomp_uop_no_pc_update;
|
||||
cir_uop_atomic <= |EXTENSION_ZCMP && !uop_clear && decomp_uop_atomic;
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
assign cir_raw = {
|
||||
buf_contents[1 * W_SLOT +: W_BUNDLE],
|
||||
buf_contents[0 * W_SLOT +: W_BUNDLE]
|
||||
};
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+478
@@ -0,0 +1,478 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2023 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Little instructions go in, big instructions come out
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_instr_decompress #(
|
||||
`include "hazard3_config.vh"
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
|
||||
input wire [31:0] instr_in,
|
||||
|
||||
output reg instr_is_32bit,
|
||||
output reg [31:0] instr_out,
|
||||
|
||||
// If instruction is a non-final uop, need to suppress PC update, and null
|
||||
// the PC offset in the mepc address in stage 3.
|
||||
output wire instr_out_is_uop,
|
||||
output wire instr_out_is_final_uop,
|
||||
output wire instr_out_uop_no_pc_update,
|
||||
// Indicate instr_out is a uop from the noninterruptible part of a uop
|
||||
// sequence. If one uop is noninterruptible, all following uops until the
|
||||
// end of the sequence are also noninterruptible.
|
||||
output wire instr_out_uop_atomic,
|
||||
// Current ucode sequence is stalled on downstream execution
|
||||
input wire instr_out_uop_stall,
|
||||
input wire instr_out_uop_clear,
|
||||
|
||||
// To regnum decoder in frontend
|
||||
output wire [3:0] df_uop_step,
|
||||
|
||||
output reg invalid
|
||||
);
|
||||
|
||||
`include "rv_opcodes.vh"
|
||||
|
||||
localparam W_REGADDR = 5;
|
||||
localparam PASSTHROUGH = ~|EXTENSION_C;
|
||||
|
||||
// Long-register formats: cr, ci, css
|
||||
// Short-register formats: ciw, cl, cs, cb, cj
|
||||
wire [W_REGADDR-1:0] rd_l = instr_in[11:7];
|
||||
wire [W_REGADDR-1:0] rs1_l = instr_in[11:7];
|
||||
wire [W_REGADDR-1:0] rs2_l = instr_in[6:2];
|
||||
wire [W_REGADDR-1:0] rd_s = {2'b01, instr_in[4:2]};
|
||||
wire [W_REGADDR-1:0] rs1_s = {2'b01, instr_in[9:7]};
|
||||
wire [W_REGADDR-1:0] rs2_s = {2'b01, instr_in[4:2]};
|
||||
|
||||
// Mapping of cx -> x immediate formats (we are *expanding* instructions, not
|
||||
// decoding them):
|
||||
|
||||
wire [31:0] imm_ci = {
|
||||
{7{instr_in[12]}},
|
||||
instr_in[6:2],
|
||||
20'h00000
|
||||
};
|
||||
|
||||
wire [31:0] imm_cj = {
|
||||
instr_in[12],
|
||||
instr_in[8],
|
||||
instr_in[10:9],
|
||||
instr_in[6],
|
||||
instr_in[7],
|
||||
instr_in[2],
|
||||
instr_in[11],
|
||||
instr_in[5:3],
|
||||
{9{instr_in[12]}},
|
||||
12'h000
|
||||
};
|
||||
|
||||
wire [31:0] imm_cb ={
|
||||
{4{instr_in[12]}},
|
||||
instr_in[6:5],
|
||||
instr_in[2],
|
||||
13'h0000,
|
||||
instr_in[11:10],
|
||||
instr_in[4:3],
|
||||
instr_in[12],
|
||||
7'h00
|
||||
};
|
||||
|
||||
wire [31:0] imm_c_lb = {
|
||||
10'h0,
|
||||
instr_in[5],
|
||||
instr_in[6],
|
||||
20'h00000
|
||||
};
|
||||
|
||||
wire [31:0] imm_c_lh = {
|
||||
10'h000,
|
||||
instr_in[5],
|
||||
1'b0,
|
||||
20'h00000
|
||||
};
|
||||
|
||||
function [31:0] rfmt_rd; input [4:0] rd; begin rfmt_rd = {20'h00000, rd, 7'h00}; end endfunction
|
||||
function [31:0] rfmt_rs1; input [4:0] rs1; begin rfmt_rs1 = {12'h000, rs1, 15'h0000}; end endfunction
|
||||
function [31:0] rfmt_rs2; input [4:0] rs2; begin rfmt_rs2 = {7'h00, rs2, 20'h00000}; end endfunction
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Push/pop and friends
|
||||
|
||||
// The longest uop sequence is a maximal cm.popretz:
|
||||
//
|
||||
// - 13x lw (counter = 0..12)
|
||||
// - 1x addi to set a0 to zero (counter = 13 ) < atomic section
|
||||
// - 1x jalr to jump through ra (counter = 14 ) < atomic section
|
||||
// - 1x addi to adjust sp (counter = 15 ) < atomic section
|
||||
|
||||
wire [3:0] uop_ctr;
|
||||
reg [3:0] uop_ctr_nxt_in_seq;
|
||||
reg in_uop_seq;
|
||||
reg uop_no_pc_update;
|
||||
|
||||
wire zcmp_is_pushpop = instr_in[12];
|
||||
wire uop_seq_end = |EXTENSION_ZCMP && (zcmp_is_pushpop ? uop_ctr == 4'hf : uop_ctr[0]);
|
||||
wire uop_atomic = |EXTENSION_ZCMP && (zcmp_is_pushpop ? uop_ctr >= 4'he : uop_ctr[0]);
|
||||
|
||||
wire [3:0] uop_ctr_nxt =
|
||||
instr_out_uop_clear ? 4'h0 :
|
||||
instr_out_uop_stall ? uop_ctr : uop_ctr_nxt_in_seq;
|
||||
|
||||
assign instr_out_is_uop = in_uop_seq;
|
||||
assign instr_out_is_final_uop = uop_seq_end;
|
||||
assign instr_out_uop_atomic = uop_atomic;
|
||||
assign instr_out_uop_no_pc_update = uop_no_pc_update;
|
||||
assign df_uop_step = uop_ctr;
|
||||
|
||||
// The offset from current sp value to the lowest-addressed saved register, +64.
|
||||
wire [3:0] zcmp_rlist = instr_in[7:4];
|
||||
wire [3:0] zcmp_n_regs = zcmp_rlist == 4'hf ? 4'hd : zcmp_rlist - 4'h3;
|
||||
wire zcmp_rlist_invalid = zcmp_rlist < 4'h4 || (|EXTENSION_E && zcmp_rlist > 4'h6);
|
||||
|
||||
wire [11:0] zcmp_stack_adj_base =
|
||||
zcmp_rlist == 4'hf ? 12'h040 :
|
||||
zcmp_rlist >= 4'hc ? 12'h030 :
|
||||
zcmp_rlist >= 4'h8 ? 12'h020 : 12'h010;
|
||||
|
||||
wire [11:0] zcmp_stack_adj = zcmp_stack_adj_base + {6'h00, instr_in[3:2], 4'h0};
|
||||
|
||||
// Note we perform all load/stores before moving the stack pointer.
|
||||
wire [11:0] zcmp_stack_lw_offset = -{6'h00, {zcmp_n_regs - uop_ctr}, 2'h0} + zcmp_stack_adj;
|
||||
wire [11:0] zcmp_stack_sw_offset = -{6'h00, {zcmp_n_regs - uop_ctr}, 2'h0};
|
||||
|
||||
wire [4:0] zcmp_ls_reg =
|
||||
uop_ctr == 4'h0 ? 5'd01 : // ra
|
||||
uop_ctr == 4'h1 ? 5'd08 : // s0
|
||||
uop_ctr == 4'h2 ? 5'd09 : // s1
|
||||
5'd15 + {1'b0, uop_ctr}; // s2-s11 (s2 == x18)
|
||||
|
||||
wire [31:0] zcmp_push_sw_instr = `RVOPC_NOZ_SW | rfmt_rs1(5'd2) | rfmt_rs2(zcmp_ls_reg) | {
|
||||
zcmp_stack_sw_offset[11:5], 13'h0000, zcmp_stack_sw_offset[4:0], 7'h00
|
||||
};
|
||||
|
||||
wire [31:0] zcmp_pop_lw_instr = `RVOPC_NOZ_LW | rfmt_rd(zcmp_ls_reg) | rfmt_rs1(5'd2)| {
|
||||
zcmp_stack_lw_offset[11:0], 20'h00000
|
||||
};
|
||||
|
||||
wire [31:0] zcmp_push_stack_adj_instr = `RVOPC_NOZ_ADDI | rfmt_rd(5'd2) | rfmt_rs1(5'd2) | {
|
||||
-zcmp_stack_adj,
|
||||
20'h00000
|
||||
};
|
||||
|
||||
wire [31:0] zcmp_pop_stack_adj_instr = `RVOPC_NOZ_ADDI | rfmt_rd(5'd2) | rfmt_rs1(5'd2) | {
|
||||
zcmp_stack_adj,
|
||||
20'h00000
|
||||
};
|
||||
|
||||
wire [4:0] zcmp_sa01_r1s = {|instr_in[9:8], ~|instr_in[9:8], instr_in[9:7]};
|
||||
wire [4:0] zcmp_sa01_r2s = {|instr_in[4:3], ~|instr_in[4:3], instr_in[4:2]};
|
||||
|
||||
wire zcmp_sa01_invalid = |EXTENSION_E && |{instr_in[9:8], instr_in[4:3]};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
generate
|
||||
if (PASSTHROUGH) begin: instr_passthrough
|
||||
always @ (*) begin
|
||||
instr_is_32bit = 1'b1;
|
||||
instr_out = instr_in;
|
||||
invalid = 1'b0;
|
||||
end
|
||||
end else begin: instr_decompress
|
||||
always @ (*) begin
|
||||
if (instr_in[1:0] == 2'b11) begin
|
||||
instr_is_32bit = 1'b1;
|
||||
instr_out = instr_in;
|
||||
invalid = 1'b0;
|
||||
in_uop_seq = 1'b0;
|
||||
uop_no_pc_update = 1'b0;
|
||||
uop_ctr_nxt_in_seq = uop_ctr;
|
||||
end else begin
|
||||
instr_is_32bit = 1'b0;
|
||||
instr_out = 32'd0;
|
||||
invalid = 1'b0;
|
||||
in_uop_seq = 1'b0;
|
||||
uop_no_pc_update = 1'b0;
|
||||
uop_ctr_nxt_in_seq = uop_ctr;
|
||||
casez (instr_in[15:0])
|
||||
`RVOPC_C_ADDI4SPN: begin
|
||||
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(rd_s) | rfmt_rs1(5'd2)
|
||||
| {2'h0, instr_in[10:7], instr_in[12:11], instr_in[5], instr_in[6], 2'b00, 20'h00000};
|
||||
invalid = ~|instr_in[12:2]; // Always-invalid all-zeroes instruction
|
||||
end
|
||||
`RVOPC_C_LW: instr_out = `RVOPC_NOZ_LW | rfmt_rd(rd_s) | rfmt_rs1(rs1_s)
|
||||
| {5'h00, instr_in[5], instr_in[12:10], instr_in[6], 2'b00, 20'h00000};
|
||||
`RVOPC_C_SW: instr_out = `RVOPC_NOZ_SW | rfmt_rs2(rs2_s) | rfmt_rs1(rs1_s)
|
||||
| {5'h00, instr_in[5], instr_in[12], 13'h0000, instr_in[11:10], instr_in[6], 2'b00, 7'h00};
|
||||
`RVOPC_C_ADDI: instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(rd_l) | rfmt_rs1(rs1_l) | imm_ci;
|
||||
`RVOPC_C_JAL: instr_out = `RVOPC_NOZ_JAL | rfmt_rd(5'd1) | imm_cj;
|
||||
`RVOPC_C_J: instr_out = `RVOPC_NOZ_JAL | rfmt_rd(5'd0) | imm_cj;
|
||||
`RVOPC_C_LI: instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(rd_l) | imm_ci;
|
||||
`RVOPC_C_LUI: begin
|
||||
if (rd_l == 5'd2) begin
|
||||
// addi16sp
|
||||
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(5'd2) | rfmt_rs1(5'd2) |
|
||||
{{3{instr_in[12]}}, instr_in[4:3], instr_in[5], instr_in[2], instr_in[6], 24'h000000};
|
||||
end else begin
|
||||
instr_out = `RVOPC_NOZ_LUI | rfmt_rd(rd_l) | {{15{instr_in[12]}}, instr_in[6:2], 12'h000};
|
||||
end
|
||||
invalid = ~|{instr_in[12], instr_in[6:2]}; // RESERVED if imm == 0
|
||||
end
|
||||
`RVOPC_C_SLLI: instr_out = `RVOPC_NOZ_SLLI | rfmt_rd(rs1_l) | rfmt_rs1(rs1_l) | imm_ci;
|
||||
`RVOPC_C_SRAI: instr_out = `RVOPC_NOZ_SRAI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | imm_ci;
|
||||
`RVOPC_C_SRLI: instr_out = `RVOPC_NOZ_SRLI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | imm_ci;
|
||||
`RVOPC_C_ANDI: instr_out = `RVOPC_NOZ_ANDI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | imm_ci;
|
||||
`RVOPC_C_AND: instr_out = `RVOPC_NOZ_AND | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
|
||||
`RVOPC_C_OR: instr_out = `RVOPC_NOZ_OR | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
|
||||
`RVOPC_C_XOR: instr_out = `RVOPC_NOZ_XOR | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
|
||||
`RVOPC_C_SUB: instr_out = `RVOPC_NOZ_SUB | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
|
||||
`RVOPC_C_ADD: begin
|
||||
if (|rs2_l) begin
|
||||
instr_out = `RVOPC_NOZ_ADD | rfmt_rd(rd_l) | rfmt_rs1(rs1_l) | rfmt_rs2(rs2_l);
|
||||
end else if (|rs1_l) begin // jalr
|
||||
instr_out = `RVOPC_NOZ_JALR | rfmt_rd(5'd1) | rfmt_rs1(rs1_l);
|
||||
end else begin // ebreak
|
||||
instr_out = `RVOPC_NOZ_EBREAK;
|
||||
end
|
||||
end
|
||||
`RVOPC_C_MV: begin
|
||||
if (|rs2_l) begin // mv
|
||||
instr_out = `RVOPC_NOZ_ADD | rfmt_rd(rd_l) | rfmt_rs2(rs2_l);
|
||||
end else begin // jr
|
||||
instr_out = `RVOPC_NOZ_JALR | rfmt_rs1(rs1_l);
|
||||
invalid = ~|rs1_l; // RESERVED
|
||||
end
|
||||
end
|
||||
`RVOPC_C_LWSP: begin
|
||||
instr_out = `RVOPC_NOZ_LW | rfmt_rd(rd_l) | rfmt_rs1(5'd2) |
|
||||
{4'h0, instr_in[3:2], instr_in[12], instr_in[6:4], 2'b00, 20'h00000};
|
||||
invalid = ~|rd_l; // RESERVED
|
||||
end
|
||||
`RVOPC_C_SWSP: instr_out = `RVOPC_NOZ_SW | rfmt_rs2(rs2_l) | rfmt_rs1(5'd2)
|
||||
| {4'h0, instr_in[8:7], instr_in[12], 13'h0000, instr_in[11:9], 2'b00, 7'h00};
|
||||
`RVOPC_C_BEQZ: instr_out = `RVOPC_NOZ_BEQ | rfmt_rs1(rs1_s) | imm_cb;
|
||||
`RVOPC_C_BNEZ: instr_out = `RVOPC_NOZ_BNE | rfmt_rs1(rs1_s) | imm_cb;
|
||||
|
||||
// Optional Zcb instructions:
|
||||
`RVOPC_C_LBU: begin
|
||||
instr_out = `RVOPC_NOZ_LBU | rfmt_rd(rd_s) | rfmt_rs1(rs1_s) | imm_c_lb;
|
||||
invalid = ~|EXTENSION_ZCB;
|
||||
end
|
||||
`RVOPC_C_LHU: begin
|
||||
instr_out = `RVOPC_NOZ_LHU | rfmt_rd(rd_s) | rfmt_rs1(rs1_s) | imm_c_lh;
|
||||
invalid = ~|EXTENSION_ZCB;
|
||||
end
|
||||
`RVOPC_C_LH: begin
|
||||
instr_out = `RVOPC_NOZ_LH | rfmt_rd(rd_s) | rfmt_rs1(rs1_s) | imm_c_lh;
|
||||
invalid = ~|EXTENSION_ZCB;
|
||||
end
|
||||
`RVOPC_C_SB: begin
|
||||
instr_out = `RVOPC_NOZ_SB | rfmt_rs2(rd_s) | rfmt_rs1(rs1_s) | imm_c_lb >> 13;
|
||||
invalid = ~|EXTENSION_ZCB;
|
||||
end
|
||||
`RVOPC_C_SH: begin
|
||||
instr_out = `RVOPC_NOZ_SH | rfmt_rs2(rd_s) | rfmt_rs1(rs1_s) | imm_c_lh >> 13;
|
||||
invalid = ~|EXTENSION_ZCB;
|
||||
end
|
||||
`RVOPC_C_ZEXT_B: begin
|
||||
instr_out = `RVOPC_NOZ_ANDI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | 32'h0ff00000;
|
||||
invalid = ~|EXTENSION_ZCB;
|
||||
end
|
||||
`RVOPC_C_SEXT_B: begin
|
||||
instr_out = `RVOPC_NOZ_SEXT_B | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s);
|
||||
invalid = ~|EXTENSION_ZCB || ~|EXTENSION_ZBB;
|
||||
end
|
||||
`RVOPC_C_ZEXT_H: begin
|
||||
instr_out = `RVOPC_NOZ_ZEXT_H | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s);
|
||||
invalid = ~|EXTENSION_ZCB || ~|EXTENSION_ZBB;
|
||||
end
|
||||
`RVOPC_C_SEXT_H: begin
|
||||
instr_out = `RVOPC_NOZ_SEXT_H | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s);
|
||||
invalid = ~|EXTENSION_ZCB || ~|EXTENSION_ZBB;
|
||||
end
|
||||
`RVOPC_C_NOT: begin
|
||||
instr_out = `RVOPC_NOZ_XORI | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | 32'hfff00000;
|
||||
invalid = ~|EXTENSION_ZCB;
|
||||
end
|
||||
`RVOPC_C_MUL: begin
|
||||
instr_out = `RVOPC_NOZ_MUL | rfmt_rd(rs1_s) | rfmt_rs1(rs1_s) | rfmt_rs2(rs2_s);
|
||||
invalid = ~|EXTENSION_ZCB || ~|EXTENSION_M;
|
||||
end
|
||||
|
||||
// Optional Zclsd instructions:
|
||||
`RVOPC_C_LD: begin
|
||||
instr_out = `RVOPC_NOZ_LD | rfmt_rd(rd_s) | rfmt_rs1(rs1_s)
|
||||
| {4'h0, instr_in[6:5], instr_in[12:10], 3'b000, 20'h00000};
|
||||
invalid = ~|EXTENSION_ZILSD || ~|EXTENSION_ZCLSD;
|
||||
end
|
||||
`RVOPC_C_SD: begin
|
||||
instr_out = `RVOPC_NOZ_SD | rfmt_rs2(rs2_s) | rfmt_rs1(rs1_s)
|
||||
| {4'h0, instr_in[6:5], instr_in[12], 13'h0000, instr_in[11:10], 3'b000, 7'h00};
|
||||
invalid = ~|EXTENSION_ZILSD || ~|EXTENSION_ZCLSD;
|
||||
end
|
||||
`RVOPC_C_LDSP: begin
|
||||
instr_out = `RVOPC_NOZ_LD | rfmt_rd(rd_l) | rfmt_rs1(5'd2) |
|
||||
{3'h0, instr_in[4:2], instr_in[12], instr_in[6:5], 3'b000, 20'h00000};
|
||||
invalid = ~|EXTENSION_ZILSD || ~|EXTENSION_ZCLSD || ~|rd_l; // RESERVED
|
||||
end
|
||||
`RVOPC_C_SDSP: begin
|
||||
instr_out = `RVOPC_NOZ_SD | rfmt_rs2(rs2_l) | rfmt_rs1(5'd2)
|
||||
| {3'h0, instr_in[9:7], instr_in[12], 13'h0000, instr_in[11:10], 3'b000, 7'h00};
|
||||
invalid = ~|EXTENSION_ZILSD || ~|EXTENSION_ZCLSD;
|
||||
end
|
||||
|
||||
// Optional Zcmp instructions:
|
||||
`RVOPC_CM_PUSH: if (~|EXTENSION_ZCMP || zcmp_rlist_invalid) begin
|
||||
invalid = 1'b1;
|
||||
end else if (uop_ctr == 4'hf) begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = 4'h0;
|
||||
instr_out = zcmp_push_stack_adj_instr;
|
||||
end else begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
|
||||
instr_out = zcmp_push_sw_instr;
|
||||
uop_no_pc_update = 1'b1;
|
||||
if (uop_ctr_nxt_in_seq == zcmp_n_regs) begin
|
||||
uop_ctr_nxt_in_seq = 4'hf;
|
||||
end
|
||||
end
|
||||
|
||||
`RVOPC_CM_POP: if (~|EXTENSION_ZCMP || zcmp_rlist_invalid) begin
|
||||
invalid = 1'b1;
|
||||
end else if (uop_ctr == 4'hf) begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = 4'h0;
|
||||
instr_out = zcmp_pop_stack_adj_instr;
|
||||
end else begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
|
||||
uop_no_pc_update = 1'b1;
|
||||
instr_out = zcmp_pop_lw_instr;
|
||||
if (uop_ctr_nxt_in_seq == zcmp_n_regs) begin
|
||||
uop_ctr_nxt_in_seq = 4'hf;
|
||||
end
|
||||
end
|
||||
|
||||
`RVOPC_CM_POPRET: if (~|EXTENSION_ZCMP || zcmp_rlist_invalid) begin
|
||||
invalid = 1'b1;
|
||||
end else if (uop_ctr == 4'he) begin
|
||||
// Note although this is only the first instruction in the uninterruptible sequence,
|
||||
// we mark this instruction as uninterruptible: there is some special case logic to
|
||||
// allow this jump to execute without flushing the final stack adjust uop, which can
|
||||
// cause the wrong exception PC to be sampled if this uop is interrupted.
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
|
||||
instr_out = `RVOPC_NOZ_JALR | rfmt_rs1(5'd1);
|
||||
end else if (uop_ctr == 4'hf) begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = 4'h0;
|
||||
uop_no_pc_update = 1'b1;
|
||||
instr_out = zcmp_pop_stack_adj_instr;
|
||||
end else begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
|
||||
instr_out = zcmp_pop_lw_instr;
|
||||
uop_no_pc_update = 1'b1;
|
||||
if (uop_ctr_nxt_in_seq == zcmp_n_regs) begin
|
||||
uop_ctr_nxt_in_seq = 4'he;
|
||||
end
|
||||
end
|
||||
|
||||
`RVOPC_CM_POPRETZ: if (~|EXTENSION_ZCMP || zcmp_rlist_invalid) begin
|
||||
invalid = 1'b1;
|
||||
end else if (uop_ctr == 4'hd) begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
|
||||
uop_no_pc_update = 1'b1;
|
||||
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(5'd10); // li a0, 0
|
||||
end else if (uop_ctr == 4'he) begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
|
||||
instr_out = `RVOPC_NOZ_JALR | rfmt_rs1(5'd1);
|
||||
end else if (uop_ctr == 4'hf) begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = 4'h0;
|
||||
uop_no_pc_update = 1'b1;
|
||||
instr_out = zcmp_pop_stack_adj_instr;
|
||||
end else begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
|
||||
uop_no_pc_update = 1'b1;
|
||||
instr_out = zcmp_pop_lw_instr;
|
||||
if (uop_ctr_nxt_in_seq == zcmp_n_regs) begin
|
||||
uop_ctr_nxt_in_seq = 4'hd;
|
||||
end
|
||||
end
|
||||
|
||||
`RVOPC_CM_MVSA01: if (~|EXTENSION_ZCMP || zcmp_sa01_invalid) begin
|
||||
invalid = 1'b1;
|
||||
end else if (uop_ctr == 4'h0) begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
|
||||
uop_no_pc_update = 1'b1;
|
||||
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(zcmp_sa01_r1s) | rfmt_rs1(5'd10);
|
||||
end else begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = 4'h0;
|
||||
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(zcmp_sa01_r2s) | rfmt_rs1(5'd11);
|
||||
end
|
||||
|
||||
`RVOPC_CM_MVA01S: if (~|EXTENSION_ZCMP || zcmp_sa01_invalid) begin
|
||||
invalid = 1'b1;
|
||||
end else if (uop_ctr == 4'h0) begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = uop_ctr + 4'h1;
|
||||
uop_no_pc_update = 1'b1;
|
||||
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(5'd10) | rfmt_rs1(zcmp_sa01_r1s);
|
||||
end else begin
|
||||
in_uop_seq = 1'b1;
|
||||
uop_ctr_nxt_in_seq = 4'h0;
|
||||
instr_out = `RVOPC_NOZ_ADDI | rfmt_rd(5'd11) | rfmt_rs1(zcmp_sa01_r2s);
|
||||
end
|
||||
|
||||
default: invalid = 1'b1;
|
||||
endcase
|
||||
end
|
||||
end
|
||||
end
|
||||
endgenerate
|
||||
|
||||
generate
|
||||
if (EXTENSION_ZCMP) begin: have_uop_ctr
|
||||
reg [3:0] uop_ctr_r;
|
||||
assign uop_ctr = uop_ctr_r;
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
uop_ctr_r <= 4'h0;
|
||||
end else begin
|
||||
uop_ctr_r <= uop_ctr_nxt;
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
assert(in_uop_seq || uop_ctr_r == 4'h0);
|
||||
assert(in_uop_seq || zcmp_ls_reg == 5'h01);
|
||||
assert(in_uop_seq || !uop_atomic);
|
||||
assert(in_uop_seq || !uop_no_pc_update);
|
||||
if (uop_seq_end) begin
|
||||
assert(in_uop_seq);
|
||||
assert(instr_out_uop_stall || uop_ctr_nxt == 4'h0);
|
||||
end
|
||||
`endif
|
||||
end
|
||||
end
|
||||
end else begin: no_uop_ctr
|
||||
assign uop_ctr = 4'h0;
|
||||
end
|
||||
endgenerate
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+327
@@ -0,0 +1,327 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
`default_nettype none
|
||||
|
||||
// Hazard3 interrupt controller. Support for up to 512 external interrupt
|
||||
// lines, with up to 16 levels of preemption.
|
||||
|
||||
module hazard3_irq_ctrl #(
|
||||
`include "hazard3_config.vh"
|
||||
) (
|
||||
input wire clk,
|
||||
input wire clk_always_on,
|
||||
input wire rst_n,
|
||||
|
||||
// CSR interface
|
||||
input wire [11:0] addr,
|
||||
input wire [1:0] wtype,
|
||||
input wire wen_m_mode,
|
||||
input wire ren_m_mode,
|
||||
input wire [W_DATA-1:0] wdata_raw,
|
||||
input wire [W_DATA-1:0] wdata,
|
||||
output reg [W_DATA-1:0] rdata,
|
||||
|
||||
// Trap entry/exit signals for context update
|
||||
input wire trapreg_update_enter,
|
||||
input wire trapreg_update_exit,
|
||||
input wire trap_entry_is_eirq,
|
||||
|
||||
// Interface for clearing and saving mie.mtie/msie via meicontext
|
||||
output wire meicontext_clearts,
|
||||
input wire mie_mtie,
|
||||
input wire mie_msie,
|
||||
|
||||
// External IRQ inputs:
|
||||
input wire [NUM_IRQS-1:0] irq,
|
||||
|
||||
// mip.meip:
|
||||
output wire external_irq_pending
|
||||
);
|
||||
|
||||
`include "hazard3_ops.vh"
|
||||
`include "hazard3_csr_addr.vh"
|
||||
|
||||
localparam MAX_IRQS = 512;
|
||||
localparam [3:0] IRQ_PRIORITY_MASK = ~(4'hf >> IRQ_PRIORITY_BITS);
|
||||
localparam W_IRQ_INDEX = $clog2(MAX_IRQS);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// IRQ input flops
|
||||
|
||||
// Register external IRQ signals (mainly to avoid a through-path from IRQs to
|
||||
// bus request signals). Always clocked, as it's used to generate a wakeup.
|
||||
// Input registers can be removed on a per-IRQ basis, but this should be done
|
||||
// with care as it does create a through-path from the IRQ to the bus.
|
||||
|
||||
wire [NUM_IRQS-1:0] irq_r;
|
||||
|
||||
genvar g;
|
||||
generate
|
||||
for (g = 0; g < NUM_IRQS; g = g + 1) begin: irq_reg_loop
|
||||
if (IRQ_INPUT_BYPASS[g]) begin: no_reg
|
||||
assign irq_r[g] = irq[g];
|
||||
end else begin: have_reg
|
||||
reg q;
|
||||
always @ (posedge clk_always_on or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
q <= 1'b0;
|
||||
end else begin
|
||||
q <= irq[g];
|
||||
end
|
||||
end
|
||||
assign irq_r[g] = q;
|
||||
end
|
||||
end
|
||||
endgenerate
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// CSR write
|
||||
|
||||
// Assigned later:
|
||||
wire [W_IRQ_INDEX-1:0] meinext_irq;
|
||||
wire meinext_noirq;
|
||||
reg [3:0] eirq_highest_priority;
|
||||
|
||||
// Interrupt array registers:
|
||||
reg [NUM_IRQS-1:0] meiea;
|
||||
reg [NUM_IRQS-1:0] meifa;
|
||||
reg [4*NUM_IRQS-1:0] meipra;
|
||||
|
||||
// Padded vectors for CSR readout
|
||||
wire [MAX_IRQS-1:0] meiea_rdata = {{MAX_IRQS-NUM_IRQS{1'b0}}, meiea};
|
||||
wire [MAX_IRQS-1:0] meifa_rdata = {{MAX_IRQS-NUM_IRQS{1'b0}}, meifa};
|
||||
wire [4*MAX_IRQS-1:0] meipra_rdata = {{4*(MAX_IRQS-NUM_IRQS){1'b0}}, meipra};
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin: update_irq_reg_arrays
|
||||
reg signed [31:0] i;
|
||||
if (!rst_n) begin
|
||||
meiea <= {NUM_IRQS{1'b0}};
|
||||
meifa <= {NUM_IRQS{1'b0}};
|
||||
meipra <= {4*NUM_IRQS{1'b0}};
|
||||
end else begin
|
||||
for (i = 0; i < NUM_IRQS; i = i + 1) begin
|
||||
// CSR write update. Note raw wdata is used for array indexing --
|
||||
// necessary for correctness, and also avoid a loop with rdata.
|
||||
if (wen_m_mode && addr == MEIEA && $signed(wdata_raw[4:0]) == i[W_IRQ_INDEX-1:4]) begin
|
||||
meiea[i] <= wdata[16 + (i % 16)];
|
||||
end
|
||||
if (wen_m_mode && addr == MEIFA && $signed(wdata_raw[4:0]) == i[W_IRQ_INDEX-1:4]) begin
|
||||
meifa[i] <= wdata[16 + (i % 16)];
|
||||
end
|
||||
if (wen_m_mode && addr == MEIPRA && $signed(wdata_raw[6:0]) == i[W_IRQ_INDEX-1:2]) begin
|
||||
meipra[4 * i +: 4] <= wdata[16 + 4 * (i % 4) +: 4] & IRQ_PRIORITY_MASK;
|
||||
end
|
||||
// Clear IRQ force when the corresponding IRQ is sampled from meinext
|
||||
// (so that an IRQ can be posted *once* without modifying the ISR source)
|
||||
if (meinext_irq == i[W_IRQ_INDEX-1:0] && ren_m_mode && addr == MEINEXT && !meinext_noirq) begin
|
||||
meifa[i[$clog2(NUM_IRQS)-1:0]] <= 1'b0;
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
reg [3:0] meicontext_pppreempt;
|
||||
reg [3:0] meicontext_ppreempt;
|
||||
reg [4:0] meicontext_preempt;
|
||||
reg meicontext_noirq;
|
||||
reg [W_IRQ_INDEX-1:0] meicontext_irq;
|
||||
reg meicontext_mreteirq;
|
||||
|
||||
wire [4:0] preempt_level_next = meinext_noirq ? 5'h10 : (
|
||||
(5'd1 << (4 - IRQ_PRIORITY_BITS)) + {1'b0, eirq_highest_priority}
|
||||
);
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
meicontext_pppreempt <= 4'h0;
|
||||
meicontext_ppreempt <= 4'h0;
|
||||
meicontext_preempt <= 5'h0;
|
||||
meicontext_noirq <= 1'b1;
|
||||
meicontext_irq <= {W_IRQ_INDEX{1'b0}};
|
||||
meicontext_mreteirq <= 1'b0;
|
||||
end else if (trapreg_update_enter) begin
|
||||
if (trap_entry_is_eirq) begin
|
||||
// Priority save. Note the MSB of preempt needn't be saved since,
|
||||
// when it is set, preemption is impossible, so we won't be here.
|
||||
meicontext_pppreempt <= meicontext_ppreempt & IRQ_PRIORITY_MASK;
|
||||
meicontext_ppreempt <= meicontext_preempt[3:0] & IRQ_PRIORITY_MASK;
|
||||
// Setting preempt isn't strictly necessary, since an updating read
|
||||
// of meinext ought to be performed before re-enabling IRQs via
|
||||
// mstatus.mie, but it seems the least surprising thing to do:
|
||||
meicontext_preempt <= preempt_level_next & {1'b1, IRQ_PRIORITY_MASK};
|
||||
meicontext_mreteirq <= 1'b1;
|
||||
end else begin
|
||||
meicontext_mreteirq <= 1'b0;
|
||||
end
|
||||
end else if (trapreg_update_exit) begin
|
||||
meicontext_mreteirq <= 1'b0;
|
||||
if (meicontext_mreteirq) begin
|
||||
// Priority restore
|
||||
meicontext_pppreempt <= 4'h0;
|
||||
meicontext_ppreempt <= meicontext_pppreempt & IRQ_PRIORITY_MASK;
|
||||
meicontext_preempt <= {1'b0, meicontext_ppreempt & IRQ_PRIORITY_MASK};
|
||||
end
|
||||
end else if (wen_m_mode && addr == MEICONTEXT) begin
|
||||
meicontext_pppreempt <= wdata[31:28] & IRQ_PRIORITY_MASK;
|
||||
meicontext_ppreempt <= wdata[27:24] & IRQ_PRIORITY_MASK;
|
||||
meicontext_preempt <= wdata[20:16] & {1'b1, IRQ_PRIORITY_MASK};
|
||||
meicontext_noirq <= wdata[15];
|
||||
meicontext_irq <= wdata[12:4];
|
||||
meicontext_mreteirq <= wdata[0];
|
||||
end else if (wen_m_mode && addr == MEINEXT && wdata[0]) begin
|
||||
// Interrupt has been sampled, with the update request set, so update
|
||||
// the context (including preemption level) appropriately.
|
||||
meicontext_preempt <= preempt_level_next & {1'b1, IRQ_PRIORITY_MASK};
|
||||
meicontext_noirq <= meinext_noirq;
|
||||
meicontext_irq <= meinext_irq;
|
||||
end
|
||||
end
|
||||
|
||||
assign meicontext_clearts = wen_m_mode && wtype != CSR_WTYPE_C && addr == MEICONTEXT && wdata_raw[1];
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// External interrupt logic
|
||||
|
||||
// Trap request is asserted when there is an interrupt at or above our current
|
||||
// preemption level. meinext displays interrupts at or above our *previous*
|
||||
// preemption level: this masking helps avoid re-taking IRQs in frames that you
|
||||
// have preempted.
|
||||
|
||||
wire [NUM_IRQS-1:0] meipa = irq_r | meifa;
|
||||
wire [MAX_IRQS-1:0] meipa_rdata = {{MAX_IRQS-NUM_IRQS{1'b0}}, meipa};
|
||||
|
||||
reg [NUM_IRQS-1:0] eirq_active_above_preempt;
|
||||
reg [NUM_IRQS-1:0] eirq_active_above_ppreempt;
|
||||
|
||||
always @ (*) begin: eirq_compare
|
||||
integer i;
|
||||
for (i = 0; i < NUM_IRQS; i = i + 1) begin
|
||||
eirq_active_above_preempt[i] = meipa[i] && meiea[i] && {1'b0, meipra[i * 4 +: 4]} >= meicontext_preempt;
|
||||
eirq_active_above_ppreempt[i] = meipa[i] && meiea[i] && meipra[i * 4 +: 4] >= meicontext_ppreempt;
|
||||
end
|
||||
end
|
||||
|
||||
assign external_irq_pending = |eirq_active_above_preempt;
|
||||
assign meinext_noirq = ~|eirq_active_above_ppreempt;
|
||||
|
||||
// Two things remaining to calculate:
|
||||
//
|
||||
// - What is the IRQ number of the highest-priority pending IRQ that is above
|
||||
// meicontext.ppreempt
|
||||
// - What is the priority of that IRQ
|
||||
//
|
||||
// In the second case we can relax the calculation to ignore ppreempt, since it
|
||||
// only needs to be valid if such an IRQ exists. Currently we choose to reuse
|
||||
// the same priority selector (possibly longer critpath while saving area), but
|
||||
// we could use a second priority selector that ignores ppreempt masking.
|
||||
|
||||
wire [NUM_IRQS-1:0] highest_eirq_onehot;
|
||||
wire [W_IRQ_INDEX-1:0] meinext_irq_unmasked;
|
||||
|
||||
hazard3_onehot_priority_dynamic #(
|
||||
.W_REQ (NUM_IRQS),
|
||||
.N_PRIORITIES (16),
|
||||
.PRIORITY_HIGHEST_WINS (1),
|
||||
.TIEBREAK_HIGHEST_WINS (0)
|
||||
) eirq_priority_u (
|
||||
.pri (meipra[4*NUM_IRQS-1:0] & {NUM_IRQS{IRQ_PRIORITY_MASK}}),
|
||||
.req (eirq_active_above_ppreempt),
|
||||
.gnt (highest_eirq_onehot)
|
||||
);
|
||||
|
||||
always @ (*) begin: get_highest_eirq_priority
|
||||
integer i;
|
||||
eirq_highest_priority = 4'h0;
|
||||
for (i = 0; i < NUM_IRQS; i = i + 1) begin
|
||||
eirq_highest_priority = eirq_highest_priority | (
|
||||
meipra[4 * i +: 4] & {4{highest_eirq_onehot[i]}}
|
||||
);
|
||||
end
|
||||
end
|
||||
|
||||
wire [$clog2(NUM_IRQS)-1:0] meinext_irq_unmasked_nopad;
|
||||
|
||||
hazard3_onehot_encode #(
|
||||
.W_REQ (NUM_IRQS)
|
||||
) eirq_encode_u (
|
||||
.req (highest_eirq_onehot),
|
||||
.gnt (meinext_irq_unmasked_nopad)
|
||||
);
|
||||
|
||||
generate
|
||||
if ($clog2(NUM_IRQS) == $clog2(MAX_IRQS)) begin: encode_eirq_no_padding
|
||||
assign meinext_irq_unmasked = meinext_irq_unmasked_nopad;
|
||||
end else begin: encode_eirq_padded
|
||||
assign meinext_irq_unmasked = {
|
||||
{$clog2(MAX_IRQS) - $clog2(NUM_IRQS){1'b0}},
|
||||
meinext_irq_unmasked_nopad
|
||||
};
|
||||
end
|
||||
endgenerate
|
||||
|
||||
// It is unnecessary to mask meinext_irq based on meinext_noirq because:
|
||||
// - The value of the CSR field is unimportant when noirq is set
|
||||
// - There are no IRQ inputs to the priority selector when there
|
||||
// are no IRQs, so result is already 0.
|
||||
assign meinext_irq = meinext_irq_unmasked;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// CSR read
|
||||
|
||||
always @ (*) begin
|
||||
rdata = {W_DATA{1'b0}};
|
||||
case (addr)
|
||||
|
||||
MEIEA: rdata = {
|
||||
meiea_rdata[wdata_raw[4:0] * 16 +: 16],
|
||||
16'h0
|
||||
};
|
||||
|
||||
MEIPA: rdata = {
|
||||
meipa_rdata[wdata_raw[4:0] * 16 +: 16],
|
||||
16'h0
|
||||
};
|
||||
|
||||
MEIFA: rdata = {
|
||||
meifa_rdata[wdata_raw[4:0] * 16 +: 16],
|
||||
16'h0
|
||||
};
|
||||
|
||||
MEIPRA: rdata = {
|
||||
meipra_rdata[wdata_raw[6:0] * 16 +: 16],
|
||||
16'h0
|
||||
};
|
||||
|
||||
MEINEXT: rdata = {
|
||||
meinext_noirq,
|
||||
20'h0,
|
||||
meinext_irq,
|
||||
2'h0
|
||||
};
|
||||
|
||||
MEICONTEXT: rdata = {
|
||||
meicontext_pppreempt,
|
||||
meicontext_ppreempt,
|
||||
3'h0,
|
||||
meicontext_preempt,
|
||||
meicontext_noirq,
|
||||
2'h0,
|
||||
meicontext_irq,
|
||||
mie_mtie && meicontext_clearts,
|
||||
mie_msie && meicontext_clearts,
|
||||
1'b0,
|
||||
meicontext_mreteirq
|
||||
};
|
||||
|
||||
default: rdata = {W_DATA{1'b0}};
|
||||
endcase
|
||||
end
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
Vendored
+124
@@ -0,0 +1,124 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// ALU operation selectors
|
||||
|
||||
localparam ALUOP_ADD = 6'h00;
|
||||
localparam ALUOP_SUB = 6'h01;
|
||||
localparam ALUOP_LT = 6'h02;
|
||||
localparam ALUOP_LTU = 6'h04;
|
||||
localparam ALUOP_AND = 6'h06;
|
||||
localparam ALUOP_OR = 6'h07;
|
||||
localparam ALUOP_XOR = 6'h08;
|
||||
localparam ALUOP_SRL = 6'h09;
|
||||
localparam ALUOP_SRA = 6'h0a;
|
||||
localparam ALUOP_SLL = 6'h0b;
|
||||
localparam ALUOP_MULDIV = 6'h0c;
|
||||
localparam ALUOP_RS2 = 6'h0d; // differs from AND/OR/XOR in [1:0]
|
||||
// Bitmanip ALU operations (some also used by AMOs):
|
||||
localparam ALUOP_SHXADD = 6'h20;
|
||||
localparam ALUOP_CLZ = 6'h23;
|
||||
localparam ALUOP_CPOP = 6'h24;
|
||||
localparam ALUOP_CTZ = 6'h25;
|
||||
localparam ALUOP_ANDN = 6'h26; // Same LSBs as non-inverted
|
||||
localparam ALUOP_ORN = 6'h27; // Same LSBs as non-inverted
|
||||
localparam ALUOP_XNOR = 6'h28; // Same LSBs as non-inverted
|
||||
localparam ALUOP_MAX = 6'h29;
|
||||
localparam ALUOP_MAXU = 6'h2a;
|
||||
localparam ALUOP_MIN = 6'h2b;
|
||||
localparam ALUOP_MINU = 6'h2c;
|
||||
localparam ALUOP_ORC_B = 6'h2d;
|
||||
localparam ALUOP_REV8 = 6'h2e;
|
||||
localparam ALUOP_ROL = 6'h2f;
|
||||
localparam ALUOP_ROR = 6'h30;
|
||||
localparam ALUOP_SEXT_B = 6'h31;
|
||||
localparam ALUOP_SEXT_H = 6'h32;
|
||||
localparam ALUOP_ZEXT_H = 6'h33;
|
||||
|
||||
localparam ALUOP_CLMUL = 6'h34;
|
||||
|
||||
localparam ALUOP_BCLR = 6'h35;
|
||||
localparam ALUOP_BEXT = 6'h36;
|
||||
localparam ALUOP_BINV = 6'h37;
|
||||
localparam ALUOP_BSET = 6'h38;
|
||||
|
||||
localparam ALUOP_PACK = 6'h39;
|
||||
localparam ALUOP_PACKH = 6'h3a;
|
||||
localparam ALUOP_BREV8 = 6'h3b;
|
||||
localparam ALUOP_ZIP = 6'h3c;
|
||||
localparam ALUOP_UNZIP = 6'h3d;
|
||||
|
||||
localparam ALUOP_BEXTM = 6'h3e;
|
||||
|
||||
localparam ALUOP_XPERM = 6'h3f;
|
||||
|
||||
// Parameters to control ALU input muxes. Bypass mux paths are
|
||||
// controlled by X, so D has no parameters to choose these.
|
||||
|
||||
localparam ALUSRCA_RS1 = 1'h0;
|
||||
localparam ALUSRCA_PC = 1'h1;
|
||||
|
||||
localparam ALUSRCB_RS2 = 1'h0;
|
||||
localparam ALUSRCB_IMM = 1'h1;
|
||||
|
||||
localparam MEMOP_LW = 5'h00;
|
||||
localparam MEMOP_LH = 5'h01;
|
||||
localparam MEMOP_LB = 5'h02;
|
||||
localparam MEMOP_LHU = 5'h03;
|
||||
localparam MEMOP_LBU = 5'h04;
|
||||
localparam MEMOP_SW = 5'h05;
|
||||
localparam MEMOP_SH = 5'h06;
|
||||
localparam MEMOP_SB = 5'h07;
|
||||
|
||||
localparam MEMOP_LR_W = 5'h08;
|
||||
localparam MEMOP_SC_W = 5'h09;
|
||||
localparam MEMOP_AMO = 5'h0a;
|
||||
localparam MEMOP_NONE = 5'h10;
|
||||
|
||||
localparam BCOND_NEVER = 2'h0;
|
||||
localparam BCOND_ALWAYS = 2'h1;
|
||||
localparam BCOND_ZERO = 2'h2;
|
||||
localparam BCOND_NZERO = 2'h3;
|
||||
|
||||
// CSR access types
|
||||
|
||||
localparam CSR_WTYPE_W = 2'h0;
|
||||
localparam CSR_WTYPE_S = 2'h1;
|
||||
localparam CSR_WTYPE_C = 2'h2;
|
||||
|
||||
// Exceptional condition signals which travel alongside (or instead of)
|
||||
// instructions in the pipeline. These are speculative and can be flushed on
|
||||
// e.g. branch mispredict. These mostly align with mcause values.
|
||||
|
||||
localparam EXCEPT_NONE = 4'hf;
|
||||
|
||||
localparam EXCEPT_INSTR_MISALIGN = 4'h0;
|
||||
localparam EXCEPT_INSTR_FAULT = 4'h1;
|
||||
localparam EXCEPT_INSTR_ILLEGAL = 4'h2;
|
||||
localparam EXCEPT_EBREAK = 4'h3;
|
||||
localparam EXCEPT_LOAD_ALIGN = 4'h4;
|
||||
localparam EXCEPT_LOAD_FAULT = 4'h5;
|
||||
localparam EXCEPT_STORE_ALIGN = 4'h6;
|
||||
localparam EXCEPT_STORE_FAULT = 4'h7;
|
||||
localparam EXCEPT_ECALL_U = 4'h8;
|
||||
// MRET, Return from M-mode: not really an exception, but handled like one
|
||||
localparam EXCEPT_MRET = 4'ha;
|
||||
localparam EXCEPT_ECALL_M = 4'hb;
|
||||
// spare: c
|
||||
// spare: d
|
||||
// REFETCH: flush and refetch sequentially-following instructions, e.g. on
|
||||
// executing fence.i. Jumps from stage 3 to get ordering against L/S dphase.
|
||||
localparam EXCEPT_REFETCH = 4'he;
|
||||
|
||||
// Operations for M extension (these are just instr[14:12])
|
||||
|
||||
localparam M_OP_MUL = 3'h0;
|
||||
localparam M_OP_MULH = 3'h1;
|
||||
localparam M_OP_MULHSU = 3'h2;
|
||||
localparam M_OP_MULHU = 3'h3;
|
||||
localparam M_OP_DIV = 3'h4;
|
||||
localparam M_OP_DIVU = 3'h5;
|
||||
localparam M_OP_REM = 3'h6;
|
||||
localparam M_OP_REMU = 3'h7;
|
||||
Vendored
+380
@@ -0,0 +1,380 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
`default_nettype none
|
||||
|
||||
// Physical memory protection unit
|
||||
|
||||
module hazard3_pmp #(
|
||||
`include "hazard3_config.vh"
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
|
||||
// Config interface passed through CSR block
|
||||
input wire [11:0] cfg_addr,
|
||||
input wire cfg_wen,
|
||||
input wire [W_DATA-1:0] cfg_wdata,
|
||||
output reg [W_DATA-1:0] cfg_rdata,
|
||||
|
||||
// Fetch address query
|
||||
input wire [W_ADDR-1:0] i_addr,
|
||||
input wire i_m_mode,
|
||||
output wire i_kill,
|
||||
|
||||
// Load/store address query
|
||||
input wire [W_ADDR-1:0] d_addr,
|
||||
// Broken out separately for carry-save:
|
||||
input wire [W_ADDR-1:0] d_addr_addend_rs1,
|
||||
input wire [W_ADDR-1:0] d_addr_addend_imm,
|
||||
input wire [W_ADDR-1:0] d_addr_addend_lspair_offs,
|
||||
input wire d_m_mode,
|
||||
input wire d_write,
|
||||
output wire d_kill
|
||||
);
|
||||
|
||||
localparam PMP_A_NAPOT = 2'b11;
|
||||
localparam PMP_A_NA4 = 2'b10;
|
||||
localparam PMP_A_TOR = 2'b01;
|
||||
localparam PMP_A_OFF = 2'b00;
|
||||
|
||||
// Which values are supported in A field (unsupported are mapped to OFF):
|
||||
localparam [3:0] PMP_A_SUPPORTED = {
|
||||
|PMP_MATCH_NAPOT,
|
||||
|PMP_MATCH_NAPOT && PMP_GRAIN == 0,
|
||||
|PMP_MATCH_TOR,
|
||||
1'b1
|
||||
};
|
||||
|
||||
`include "hazard3_csr_addr.vh"
|
||||
|
||||
generate
|
||||
if (PMP_REGIONS == 0) begin: no_pmp
|
||||
|
||||
// This should already be stubbed out in core.v, but use a generate here too
|
||||
// so that we don't get a warning for elaborating this module with a region
|
||||
// count of 0.
|
||||
|
||||
always @ (*) cfg_rdata = {W_DATA{1'b0}};
|
||||
assign i_kill = 1'b0;
|
||||
assign d_kill = 1'b0;
|
||||
|
||||
end else begin: have_pmp
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Config registers and read/write interface
|
||||
|
||||
// Whether a region's configuration is writable; this is non-trivial when TOR
|
||||
// is supported because locking region i + 1 can also lock region i.
|
||||
wire [PMP_REGIONS-1:0] region_locked;
|
||||
|
||||
reg [PMP_REGIONS-1:0] pmpcfg_l;
|
||||
reg [1:0] pmpcfg_a [0:PMP_REGIONS-1];
|
||||
reg [PMP_REGIONS-1:0] pmpcfg_x;
|
||||
reg [PMP_REGIONS-1:0] pmpcfg_w;
|
||||
reg [PMP_REGIONS-1:0] pmpcfg_r;
|
||||
|
||||
// Address register contains bits 33:2 of the address (to support 16 GiB
|
||||
// physical address space). We don't implement bits 33 or 32.
|
||||
reg [W_ADDR-3:0] pmpaddr [0:PMP_REGIONS-1];
|
||||
|
||||
// Hazard3 extension for applying PMP regions to M-mode without locking.
|
||||
// Different from ePMP mseccfg.rlb: low-numbered regions may be locked for
|
||||
// security reasons, but higher-numbered regions should stll be available for
|
||||
// other purposes e.g. stack guarding, peripheral emulation
|
||||
reg [PMP_REGIONS-1:0] pmpcfg_m;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin: cfg_update
|
||||
reg signed [31:0] i;
|
||||
if (!rst_n) begin
|
||||
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
|
||||
pmpcfg_l[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 7] : 1'b0;
|
||||
pmpcfg_a[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 3 +: 2] : 2'h0;
|
||||
pmpcfg_x[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 2] : 1'b0;
|
||||
pmpcfg_w[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 1] : 1'b0;
|
||||
pmpcfg_r[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_CFG[8 * i + 0] : 1'b0;
|
||||
|
||||
pmpaddr[i] <= PMP_HARDWIRED[i] ? PMP_HARDWIRED_ADDR[32 * i +: 30] :
|
||||
PMP_GRAIN > 1 ? ~(~30'h0 << (PMP_GRAIN - 1)) : 30'h0;
|
||||
end
|
||||
pmpcfg_m <= {PMP_REGIONS{1'b0}};
|
||||
end else if (cfg_wen) begin
|
||||
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
|
||||
if (cfg_addr == PMPCFG0 + i[13:2] && !region_locked[i]) begin
|
||||
if (PMP_HARDWIRED[i]) begin
|
||||
// Keep tied to hardwired value (but still make the "register" sensitive to clk)
|
||||
pmpcfg_l[i] <= PMP_HARDWIRED_CFG[8 * i + 7];
|
||||
pmpcfg_a[i] <= PMP_HARDWIRED_CFG[8 * i + 3 +: 2];
|
||||
pmpcfg_x[i] <= PMP_HARDWIRED_CFG[8 * i + 2];
|
||||
pmpcfg_w[i] <= PMP_HARDWIRED_CFG[8 * i + 1];
|
||||
pmpcfg_r[i] <= PMP_HARDWIRED_CFG[8 * i + 0];
|
||||
pmpaddr[i] <= PMP_HARDWIRED_ADDR[32 * i +: 30];
|
||||
end else begin
|
||||
pmpcfg_l[i] <= cfg_wdata[i % 4 * 8 + 7];
|
||||
pmpcfg_x[i] <= cfg_wdata[i % 4 * 8 + 2];
|
||||
pmpcfg_w[i] <= cfg_wdata[i % 4 * 8 + 1];
|
||||
pmpcfg_r[i] <= cfg_wdata[i % 4 * 8 + 0];
|
||||
// Unsupported A values are mapped to OFF (it's a WARL field).
|
||||
pmpcfg_a[i] <= PMP_A_SUPPORTED[cfg_wdata[i % 4 * 8 + 3 +: 2]] ?
|
||||
cfg_wdata[i % 4 * 8 + 3 +: 2] : PMP_A_OFF;
|
||||
end
|
||||
end
|
||||
if (cfg_addr == PMPADDR0 + i[11:0] && !region_locked[i]) begin
|
||||
// This implements one bit too many when G > 0 and only
|
||||
// PMP_MATCH_TOR is enabled, however that bit is ignored for
|
||||
// both rdata and address matching, so should be trimmed.
|
||||
if (PMP_GRAIN > 1) begin
|
||||
pmpaddr[i] <= cfg_wdata[W_ADDR-3:0] | ~(~30'h0 << (PMP_GRAIN - 1));
|
||||
end else begin
|
||||
pmpaddr[i] <= cfg_wdata[W_ADDR-3:0];
|
||||
end
|
||||
end
|
||||
end
|
||||
if (cfg_addr == PMPCFGM0) begin
|
||||
pmpcfg_m <= cfg_wdata[PMP_REGIONS-1:0] & ~PMP_HARDWIRED & {PMP_REGIONS{|EXTENSION_XH3PMPM}};
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
always @ (*) begin: cfg_read
|
||||
reg signed [31:0] i;
|
||||
cfg_rdata = {W_DATA{1'b0}};
|
||||
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
|
||||
if (cfg_addr == PMPCFG0 + i[13:2]) begin
|
||||
cfg_rdata[i % 4 * 8 +: 8] = {
|
||||
pmpcfg_l[i],
|
||||
2'b00,
|
||||
pmpcfg_a[i],
|
||||
pmpcfg_x[i],
|
||||
pmpcfg_w[i],
|
||||
pmpcfg_r[i]
|
||||
};
|
||||
end else if (cfg_addr == PMPADDR0 + i[11:0]) begin
|
||||
if (PMP_GRAIN >= 2 && pmpcfg_a[i][1]) begin
|
||||
// Bits G-2:0 read back as all-ones when A is NA4 or NAPOT.
|
||||
cfg_rdata[W_ADDR-3:0] = pmpaddr[i] | ~({W_ADDR-2{1'b1}} << (PMP_GRAIN - 1));
|
||||
end else if (PMP_GRAIN >= 1 && !pmpcfg_a[i][1]) begin
|
||||
// Bits G-1:0 read back as all-zeroes when A is OFF or TOR.
|
||||
cfg_rdata[W_ADDR-3:0] = pmpaddr[i] & ({W_ADDR-2{1'b1}} << PMP_GRAIN);
|
||||
end else begin
|
||||
cfg_rdata[W_ADDR-3:0] = pmpaddr[i];
|
||||
end
|
||||
end
|
||||
end
|
||||
if (cfg_addr == PMPCFGM0) begin
|
||||
cfg_rdata = {{32-PMP_REGIONS{1'b0}}, pmpcfg_m} & {32{|EXTENSION_XH3PMPM}};
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Region locking rules
|
||||
|
||||
reg [PMP_REGIONS-1:0] pmp_region_is_tor;
|
||||
always @ (*) begin: check_region_is_tor
|
||||
integer i;
|
||||
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
|
||||
pmp_region_is_tor[i] = PMP_MATCH_TOR && pmpcfg_a[i] == PMP_A_TOR;
|
||||
end
|
||||
end
|
||||
|
||||
assign region_locked = pmpcfg_l | ((pmpcfg_l & pmp_region_is_tor) >> 1);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Match addresses against regions
|
||||
|
||||
wire [PMP_REGIONS-1:0] d_match_napot;
|
||||
wire [PMP_REGIONS-1:0] i_match_napot;
|
||||
wire [PMP_REGIONS-1:0] d_match_tor;
|
||||
wire [PMP_REGIONS-1:0] i_match_tor;
|
||||
|
||||
if (PMP_MATCH_NAPOT != 0) begin: have_napot
|
||||
|
||||
reg [PMP_REGIONS-1:0] d_match_napot_r;
|
||||
reg [PMP_REGIONS-1:0] i_match_napot_r;
|
||||
|
||||
assign d_match_napot = d_match_napot_r;
|
||||
assign i_match_napot = i_match_napot_r;
|
||||
|
||||
// Decode PMPCFGx.A and PMPADDRx into a 32-bit address mask and address
|
||||
reg [W_ADDR-1:0] match_mask [0:PMP_REGIONS-1];
|
||||
reg [W_ADDR-1:0] match_addr [0:PMP_REGIONS-1];
|
||||
|
||||
// Encoding: (noting ADDR is a 4-byte address, not a word address):
|
||||
// CFG.A | ADDR | Region size
|
||||
// ------+----------+------------
|
||||
// NA4 | y..yyyyy | 4 bytes
|
||||
// NAPOT | y..yyyy0 | 8 bytes
|
||||
// NAPOT | y..yyy01 | 16 bytes
|
||||
// NAPOT | y..yy011 | 32 bytes
|
||||
// NAPOT | y..y0111 | 64 bytes
|
||||
// etc.
|
||||
//
|
||||
// So, with the exception of NA4, the rule is to check all bits more
|
||||
// significant than the least-significant 0 bit.
|
||||
|
||||
always @ (*) begin: decode_match_mask_addr
|
||||
integer i, j;
|
||||
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
|
||||
if (!pmpcfg_a[i][0]) begin
|
||||
match_mask[i] = {{W_ADDR-2{1'b1}}, 2'b00};
|
||||
end else begin
|
||||
// Bits 1:0 are always 0. Bit 2 is 0 because NAPOT is at least 8 bytes.
|
||||
match_mask[i] = {W_ADDR{1'b0}};
|
||||
for (j = 3; j < W_ADDR; j = j + 1) begin
|
||||
match_mask[i][j] = match_mask[i][j - 1] || !pmpaddr[i][j - 3];
|
||||
end
|
||||
end
|
||||
match_addr[i] = {pmpaddr[i], 2'b00} & match_mask[i];
|
||||
end
|
||||
end
|
||||
|
||||
// We check only the least-addressed byte of each access. See later
|
||||
// comments for an argument as to why this is sufficient.
|
||||
|
||||
always @ (*) begin: check_d_match
|
||||
integer i;
|
||||
for (i = PMP_REGIONS - 1; i >= 0; i = i - 1) begin
|
||||
d_match_napot_r[i] = pmpcfg_a[i][1] &&
|
||||
(d_addr & match_mask[i]) == match_addr[i];
|
||||
i_match_napot_r[i] = pmpcfg_a[i][1] &&
|
||||
(i_addr & match_mask[i]) == match_addr[i];
|
||||
end
|
||||
end
|
||||
|
||||
end else begin: no_napot
|
||||
|
||||
assign d_match_napot = {PMP_REGIONS{1'b0}};
|
||||
assign i_match_napot = {PMP_REGIONS{1'b0}};
|
||||
|
||||
end
|
||||
|
||||
if (PMP_MATCH_TOR != 0) begin: have_tor
|
||||
|
||||
reg [PMP_REGIONS-1:0] d_match_tor_r;
|
||||
reg [PMP_REGIONS-1:0] i_match_tor_r;
|
||||
reg [W_ADDR-1:0] watermark [0:PMP_REGIONS-1];
|
||||
reg [PMP_REGIONS-1:0] d_lt;
|
||||
reg [PMP_REGIONS-1:0] i_lt;
|
||||
|
||||
assign d_match_tor = d_match_tor_r;
|
||||
assign i_match_tor = i_match_tor_r;
|
||||
|
||||
always @ (*) begin: compare
|
||||
integer i;
|
||||
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
|
||||
watermark[i] = {
|
||||
pmpaddr[i][W_ADDR-3:0] & (~30'h0 << PMP_GRAIN),
|
||||
2'b00
|
||||
};
|
||||
// Bring terms in separately to try to encourage adder merging
|
||||
d_lt[i] = (
|
||||
d_addr_addend_rs1 + d_addr_addend_imm + d_addr_addend_lspair_offs
|
||||
) < watermark[i];
|
||||
i_lt[i] = i_addr < watermark[i];
|
||||
end
|
||||
end
|
||||
|
||||
wire [PMP_REGIONS-1:0] d_prev_ge = ~(d_lt << 1);
|
||||
wire [PMP_REGIONS-1:0] i_prev_ge = ~(i_lt << 1);
|
||||
|
||||
always @ (*) begin: match
|
||||
integer i;
|
||||
for (i = 0; i < PMP_REGIONS; i = i + 1) begin
|
||||
d_match_tor_r[i] = d_lt[i] && d_prev_ge[i] && pmpcfg_a[i] == PMP_A_TOR;
|
||||
i_match_tor_r[i] = i_lt[i] && i_prev_ge[i] && pmpcfg_a[i] == PMP_A_TOR;
|
||||
end
|
||||
end
|
||||
|
||||
end else begin: no_tor
|
||||
|
||||
assign d_match_tor = {PMP_REGIONS{1'b0}};
|
||||
assign i_match_tor = {PMP_REGIONS{1'b0}};
|
||||
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Decode permissions from matches
|
||||
|
||||
// For load/stores we assume any non-naturally-aligned transfers trigger a
|
||||
// misaligned load/store/AMO exception, so we only need to decode the PMP
|
||||
// attribute for the first byte of the access. Note the spec gives us freedom
|
||||
// to report *either* a load/store/AMO access fault (mcause = 5, 7) or a
|
||||
// load/store/AMO alignment fault (mcause = 4, 6), in the case that both
|
||||
// happen, and we choose alignment fault in this case.
|
||||
|
||||
reg d_m; // Hazard3 extension (M-mode without locking)
|
||||
reg d_l;
|
||||
reg d_r;
|
||||
reg d_w;
|
||||
|
||||
always @ (*) begin: check_d_match
|
||||
integer i;
|
||||
d_m = 1'b0;
|
||||
d_l = 1'b0;
|
||||
d_r = 1'b0;
|
||||
d_w = 1'b0;
|
||||
// Lowest-numbered match wins, so work down from the top. This should be
|
||||
// inferred as a priority mux structure (cascade mux).
|
||||
for (i = PMP_REGIONS - 1; i >= 0; i = i - 1) begin
|
||||
if (d_match_napot[i] || d_match_tor[i]) begin
|
||||
d_m = pmpcfg_m[i];
|
||||
d_l = pmpcfg_l[i];
|
||||
d_r = pmpcfg_r[i];
|
||||
d_w = pmpcfg_w[i];
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// Instructions work similarly because we check *fetches*, not instructions.
|
||||
// Fetch is always word-sized word-aligned. The spec permits this:
|
||||
//
|
||||
// "On some implementations, misaligned loads, stores, and instruction fetches
|
||||
// may also be decomposed into multiple accesses, some of which may succeed
|
||||
// before an access-fault exception occurs."
|
||||
//
|
||||
// Hazard3 separately checks the naturally-aligned fetches that occur in the
|
||||
// course of fetching a non-naturally-aligned instruction. This means
|
||||
// instruction fetch spanning two different regions which both grant X
|
||||
// permission *is* permitted, unlike the RP2350 version of Hazard3.
|
||||
|
||||
reg i_m; // Hazard3 extension (M-mode without locking)
|
||||
reg i_l;
|
||||
reg i_x;
|
||||
|
||||
always @ (*) begin: check_i_match
|
||||
integer i;
|
||||
i_m = 1'b0;
|
||||
i_l = 1'b0;
|
||||
i_x = 1'b0;
|
||||
for (i = PMP_REGIONS - 1; i >= 0; i = i - 1) begin
|
||||
if (i_match_napot[i] || i_match_tor[i]) begin
|
||||
i_m = pmpcfg_m[i];
|
||||
i_l = pmpcfg_l[i];
|
||||
i_x = pmpcfg_x[i];
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Access rules
|
||||
|
||||
// M-mode gets to ignore protections, unless the lock or M-mode bit is set.
|
||||
|
||||
assign d_kill = (!d_m_mode || d_l || d_m) && (
|
||||
(!d_write && !d_r) ||
|
||||
( d_write && !d_w)
|
||||
);
|
||||
|
||||
assign i_kill = (!i_m_mode || i_l || i_m) && !i_x;
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+173
@@ -0,0 +1,173 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
`default_nettype none
|
||||
|
||||
// Wake/sleep (power) state machine for Hazard3
|
||||
|
||||
module hazard3_power_ctrl #(
|
||||
`include "hazard3_config.vh"
|
||||
) (
|
||||
input wire clk_always_on,
|
||||
input wire rst_n,
|
||||
|
||||
// 4-phase (Gray code) req/ack handshake for requesting and releasing
|
||||
// power+clock enable on non-processor hardware, e.g. the bus fabric. This
|
||||
// can also be used for an external controller to gate the processor's clk
|
||||
// input, rather than the clk_en signal below.
|
||||
output reg pwrup_req,
|
||||
input wire pwrup_ack,
|
||||
|
||||
// Top-level clock enable for an optional clock gate on the processor's clk
|
||||
// input (but not clk_always_on, which clocks this module and the IRQ input
|
||||
// flops). This allows the processor to clock-gate when sleeping. It's
|
||||
// acceptable for the clock gate cell to have one cycle of delay when
|
||||
// clk_en changes.
|
||||
output reg clk_en,
|
||||
|
||||
// Power state controls from CSRs
|
||||
input wire allow_clkgate,
|
||||
input wire allow_power_down,
|
||||
input wire allow_sleep_on_block,
|
||||
|
||||
// Signal from frontend that it has stalled against the WFI pipeline
|
||||
// stall, and we are now clear to enter a deep sleep state
|
||||
input wire frontend_pwrdown_ok,
|
||||
|
||||
input wire sleeping_on_wfi,
|
||||
input wire wfi_wakeup_req,
|
||||
input wire sleeping_on_block,
|
||||
input wire block_wakeup_req_pulse,
|
||||
output reg stall_release
|
||||
);
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Wake/sleep state machine
|
||||
|
||||
localparam W_STATE = 2;
|
||||
localparam S_AWAKE = 2'h0;
|
||||
localparam S_ENTER_ASLEEP = 2'h1;
|
||||
localparam S_ASLEEP = 2'h2;
|
||||
localparam S_ENTER_AWAKE = 2'h3;
|
||||
|
||||
reg [W_STATE-1:0] state;
|
||||
reg block_wakeup_req;
|
||||
|
||||
wire active_wake_req =
|
||||
(sleeping_on_block && (block_wakeup_req || wfi_wakeup_req)) ||
|
||||
(sleeping_on_wfi && wfi_wakeup_req);
|
||||
|
||||
// Note: we assert our power up request during reset, and *assume* that the
|
||||
// power up acknowledge is also high at reset. If this is a problem, extend
|
||||
// the core reset.
|
||||
|
||||
always @ (posedge clk_always_on or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
state <= S_AWAKE;
|
||||
pwrup_req <= 1'b1;
|
||||
clk_en <= 1'b1;
|
||||
stall_release <= 1'b0;
|
||||
end else begin
|
||||
stall_release <= 1'b0;
|
||||
case (state)
|
||||
S_AWAKE: if (sleeping_on_wfi || sleeping_on_block) begin
|
||||
if (stall_release) begin
|
||||
// The last cycle of an ongoing which we have just released. Sit
|
||||
// tight, this instruction will move down the pipeline at the
|
||||
// end of this cycle. (There is an assertion that this doesn't
|
||||
// happen twice.)
|
||||
state <= S_AWAKE;
|
||||
end else if (active_wake_req) begin
|
||||
// Skip deep sleep if it would immediately fall through.
|
||||
stall_release <= 1'b1;
|
||||
end else if ((allow_power_down || allow_clkgate) && (sleeping_on_wfi || allow_sleep_on_block)) begin
|
||||
if (frontend_pwrdown_ok) begin
|
||||
pwrup_req <= !allow_power_down;
|
||||
clk_en <= !allow_clkgate;
|
||||
state <= allow_power_down ? S_ENTER_ASLEEP : S_ASLEEP;
|
||||
end else begin
|
||||
// Stay awake until it is safe to power down (i.e. until our
|
||||
// instruction fetch goes quiet).
|
||||
state <= S_AWAKE;
|
||||
end
|
||||
end else begin
|
||||
// No power state change. Just sit with the pipeline stalled.
|
||||
state <= S_AWAKE;
|
||||
end
|
||||
end
|
||||
S_ENTER_ASLEEP: if (!pwrup_ack) begin
|
||||
state <= S_ASLEEP;
|
||||
end
|
||||
S_ASLEEP: if (active_wake_req) begin
|
||||
pwrup_req <= 1'b1;
|
||||
clk_en <= 1'b1;
|
||||
// Still go through the enter state for non-power-down wakeup, in
|
||||
// case the clock gate cell has a 1 cycle delay.
|
||||
state <= S_ENTER_AWAKE;
|
||||
end
|
||||
S_ENTER_AWAKE: if (pwrup_ack || !allow_power_down) begin
|
||||
state <= S_AWAKE;
|
||||
stall_release <= 1'b1;
|
||||
end
|
||||
default: begin
|
||||
state <= S_AWAKE;
|
||||
end
|
||||
endcase
|
||||
end
|
||||
end
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
// Regs are a workaround for the non-constant reset value issue with
|
||||
// $past() in yosys-smtbmc.
|
||||
reg past_sleeping;
|
||||
reg past_stall_release;
|
||||
always @ (posedge clk_always_on) begin
|
||||
if (!rst_n) begin
|
||||
past_sleeping <= 1'b0;
|
||||
past_stall_release <= 1'b0;
|
||||
end else begin
|
||||
past_sleeping <= sleeping_on_wfi || sleeping_on_block;
|
||||
past_stall_release <= stall_release;
|
||||
// These must always be mutually exclusive.
|
||||
assert(!(sleeping_on_wfi && sleeping_on_block));
|
||||
if (stall_release) begin
|
||||
// Presumably there was a stall which we just released
|
||||
assert(past_sleeping);
|
||||
// Presumably we are still in that stall
|
||||
assert(sleeping_on_wfi|| sleeping_on_block);
|
||||
// It takes one cycle to do a release and enter a new sleep state, so a
|
||||
// double release should be impossible.
|
||||
assert(!past_stall_release);
|
||||
end
|
||||
if (state == S_ASLEEP) begin
|
||||
assert(allow_power_down || allow_clkgate);
|
||||
end
|
||||
end
|
||||
end
|
||||
`endif
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Pulse->level for block wakeup
|
||||
|
||||
// Unblock signal is sticky: a prior unblock with no block since will cause
|
||||
// the next block to immediately fall through.
|
||||
|
||||
always @ (posedge clk_always_on or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
block_wakeup_req <= 1'b0;
|
||||
end else begin
|
||||
// Note the OR takes precedence over the AND, so we don't miss a second
|
||||
// unblock that arrives at the instant we wake up.
|
||||
block_wakeup_req <= (block_wakeup_req && !(
|
||||
sleeping_on_block && stall_release
|
||||
)) || block_wakeup_req_pulse;
|
||||
end
|
||||
end
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+84
@@ -0,0 +1,84 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Register file
|
||||
// Single write port, dual read port
|
||||
|
||||
`default_nettype none
|
||||
|
||||
module hazard3_regfile_1w2r #(
|
||||
`include "hazard3_config.vh"
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
|
||||
input wire [4:0] raddr1,
|
||||
output reg [W_DATA-1:0] rdata1,
|
||||
|
||||
input wire [4:0] raddr2,
|
||||
output reg [W_DATA-1:0] rdata2,
|
||||
|
||||
input wire [4:0] waddr,
|
||||
input wire [W_DATA-1:0] wdata,
|
||||
input wire wen
|
||||
);
|
||||
|
||||
localparam N_REGS = EXTENSION_E == 0 ? 32 : 16;
|
||||
localparam [4:0] REGNUM_MASK = {~|EXTENSION_E, 4'hf};
|
||||
|
||||
wire [4:0] raddr1_masked = raddr1 & REGNUM_MASK;
|
||||
wire [4:0] raddr2_masked = raddr2 & REGNUM_MASK;
|
||||
wire [4:0] waddr_masked = waddr & REGNUM_MASK;
|
||||
|
||||
generate
|
||||
if (RESET_REGFILE) begin: real_dualport_reset
|
||||
// This will presumably always be implemented with flops
|
||||
reg [W_DATA-1:0] mem [0:N_REGS-1];
|
||||
|
||||
integer i;
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
for (i = 0; i < N_REGS; i = i + 1) begin
|
||||
mem[i] <= {W_DATA{1'b0}};
|
||||
end
|
||||
rdata1 <= {W_DATA{1'b0}};
|
||||
rdata2 <= {W_DATA{1'b0}};
|
||||
end else begin
|
||||
if (wen) begin
|
||||
mem[waddr_masked] <= wdata;
|
||||
end
|
||||
rdata1 <= mem[raddr1_masked];
|
||||
rdata2 <= mem[raddr2_masked];
|
||||
end
|
||||
end
|
||||
end else begin: real_dualport_noreset
|
||||
// This should be inference-compatible on FPGAs with dual-port (or 1R1W) BRAMs
|
||||
`ifdef YOSYS
|
||||
`ifdef FPGA_ICE40
|
||||
// We do not require write-to-read bypass logic on the BRAM
|
||||
(* no_rw_check *)
|
||||
`endif
|
||||
`endif
|
||||
// Optionally force use of distributed RAM on Xilinx for better timing
|
||||
`ifdef HAZARD3_REGFILE_RAM_STYLE_DISTRIBUTED
|
||||
(* ram_style = "distributed" *)
|
||||
`endif
|
||||
reg [W_DATA-1:0] mem [0:N_REGS-1];
|
||||
|
||||
always @ (posedge clk) begin
|
||||
if (wen) begin
|
||||
mem[waddr_masked] <= wdata;
|
||||
end
|
||||
rdata1 <= mem[raddr1_masked];
|
||||
rdata2 <= mem[raddr2_masked];
|
||||
end
|
||||
end
|
||||
endgenerate
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+377
@@ -0,0 +1,377 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2025 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// RVFI Instrumentation
|
||||
// ----------------------------------------------------------------------------
|
||||
// To be included into hazard3_core.v for use with riscv-formal.
|
||||
// Contains some state modelling to diagnose exactly what the core is doing,
|
||||
// and report this in a way RVFI understands.
|
||||
// We consider instructions to "retire" as they cross the M/W pipe register.
|
||||
//
|
||||
// All modelling signals prefixed with rvfm (riscv-formal monitor)
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Instruction monitor
|
||||
|
||||
// Diagnose whether X, M contain valid in-flight instructions, to produce
|
||||
// rvfi_valid signal.
|
||||
|
||||
wire rvfm_x_valid = fd_cir_vld >= 2 || (fd_cir_vld >= 1 && fd_cir_raw[1:0] != 2'b11);
|
||||
|
||||
reg rvfm_m_valid;
|
||||
reg [31:0] rvfm_m_instr;
|
||||
|
||||
wire rvfm_m_trap = xm_except != EXCEPT_NONE && xm_except != EXCEPT_MRET && m_trap_enter_rdy;
|
||||
|
||||
reg rvfi_valid_r;
|
||||
reg [31:0] rvfi_insn_r;
|
||||
reg rvfi_trap_r;
|
||||
|
||||
assign rvfi_valid = rvfi_valid_r;
|
||||
assign rvfi_insn = rvfi_insn_r;
|
||||
assign rvfi_trap = rvfi_trap_r;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
rvfm_m_valid <= 1'b0;
|
||||
rvfi_valid_r <= 1'b0;
|
||||
rvfi_trap_r <= 1'b0;
|
||||
rvfi_insn_r <= 32'h0;
|
||||
end else begin
|
||||
if (!x_stall) begin
|
||||
// X instruction squashed by any trap, as it's in the branch
|
||||
// shadow.
|
||||
rvfm_m_valid <= |df_cir_use && !m_trap_enter_vld;
|
||||
rvfm_m_instr <= {fd_cir_raw[31:16] & {16{df_cir_use[1]}}, fd_cir_raw[15:0]};
|
||||
end else if (!m_stall) begin
|
||||
rvfm_m_valid <= 1'b0;
|
||||
end
|
||||
rvfi_valid_r <= rvfm_m_valid && !m_stall;
|
||||
// Instructions which experienced fetch faults are reported as
|
||||
// all-zeroes, per riscv-formal docs.
|
||||
rvfi_insn_r <= rvfm_m_instr & {32{
|
||||
xm_except != EXCEPT_INSTR_FAULT &&
|
||||
xm_except != EXCEPT_INSTR_MISALIGN
|
||||
}};
|
||||
rvfi_trap_r <= rvfm_m_trap;
|
||||
end
|
||||
end
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
always @ (posedge clk) if (rst_n) begin
|
||||
// Sanity checks for above
|
||||
if (d_rd != 5'h0)
|
||||
assert(rvfm_x_valid);
|
||||
if (xm_rd != 5'h0)
|
||||
assert(rvfm_m_valid);
|
||||
end
|
||||
`endif
|
||||
|
||||
// Track whether an instruction is the first of an interrupt or exception;
|
||||
// when a trap happens, a flag is installed in stage X, and once a new
|
||||
// instruction arrives the flag travels alongside it down to the RVFI port.
|
||||
reg rvfm_x_intr;
|
||||
reg rvfm_m_intr;
|
||||
reg rvfi_intr_r;
|
||||
always @ (posedge clk) begin
|
||||
if (!rst_n) begin
|
||||
rvfm_x_intr <= 1'b0;
|
||||
rvfm_m_intr <= 1'b0;
|
||||
rvfi_intr_r <= 1'b0;
|
||||
end else begin
|
||||
rvfm_x_intr <= (rvfm_x_intr && (x_stall || d_starved || fd_cir_uop_nonfinal || df_lspair_phase_next)) ||
|
||||
(m_trap_enter_vld && m_trap_enter_rdy);
|
||||
if (!x_stall) begin
|
||||
rvfm_m_intr <= rvfm_x_intr;
|
||||
end
|
||||
if (!m_stall) begin
|
||||
rvfi_intr_r <= rvfm_m_intr;
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
// Hazard3 is an in-order core:
|
||||
reg [63:0] rvfm_retire_ctr;
|
||||
assign rvfi_order = rvfm_retire_ctr;
|
||||
always @ (posedge clk or negedge rst_n)
|
||||
if (!rst_n)
|
||||
rvfm_retire_ctr <= 0;
|
||||
else if (rvfi_valid)
|
||||
rvfm_retire_ctr <= rvfm_retire_ctr + 1;
|
||||
|
||||
assign rvfi_intr = rvfi_intr_r && rvfi_valid;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// PC and jump monitor
|
||||
|
||||
reg [31:0] rvfm_xm_pc;
|
||||
reg [31:0] rvfm_xm_pc_next;
|
||||
|
||||
// Record a jump target that was issued while stalled
|
||||
reg rvfm_x_saw_f_jump;
|
||||
reg [31:0] rvfm_x_saw_f_jump_target;
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
rvfm_x_saw_f_jump <= 1'b0;
|
||||
rvfm_x_saw_f_jump_target <= 32'd0;
|
||||
end else if (f_jump_now && !(m_trap_enter_vld && m_trap_enter_rdy) && (x_stall || (fd_cir_is_uop && fd_cir_uop_nonfinal))) begin
|
||||
// Record fetch address issued by instruction. Note this case is gated
|
||||
// on m_trap_enter_vld && m_trap_enter_rdy (not just vld). If
|
||||
// !m_trap_enter_rdy it is still possible to get a new fetch address
|
||||
// in the following case:
|
||||
//
|
||||
// * M instr is a load/store (in dphase)
|
||||
//
|
||||
// * IRQ is asserted, but blocked by the load/store dphase to avoid
|
||||
// trashing exception PC of a potential data-phase bus fault
|
||||
//
|
||||
// * X instr is a jump, which stalls due to the IRQ assertion
|
||||
//
|
||||
// * X instr's PC goes through to frontend during stall (and would be
|
||||
// flushed if the IRQ went through) because stall cannot gate fetch
|
||||
// address request to avoid AHB through-path.
|
||||
//
|
||||
// * IRQ's trap address is not immediately accepted by frontend due to
|
||||
// address-phase stall on issuing jump instr's address
|
||||
//
|
||||
// * IRQ deasserts on the next cycle, so its trap address is not accepted.
|
||||
rvfm_x_saw_f_jump <= 1'b1;
|
||||
rvfm_x_saw_f_jump_target <= f_jump_target;
|
||||
end else if (!x_stall && !(fd_cir_is_uop && fd_cir_uop_nonfinal)) begin
|
||||
rvfm_x_saw_f_jump <= 1'b0;
|
||||
end else if (m_trap_enter_vld && m_trap_enter_rdy) begin
|
||||
// E.g. trap during uop sequence
|
||||
rvfm_x_saw_f_jump <= 1'b0;
|
||||
end
|
||||
end
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
rvfm_xm_pc <= 0;
|
||||
rvfm_xm_pc_next <= 0;
|
||||
end else begin
|
||||
if (!x_stall) begin
|
||||
// For cm.popret and cm.popretz the PC actually changes on the
|
||||
// penultimate uop; ignore this and retain the initial PC. Abuse
|
||||
// knowledge that the PC update is always in the atomic section,
|
||||
// and the first instruction is always interruptible.
|
||||
if (!(fd_cir_is_uop && fd_cir_uop_atomic)) begin
|
||||
rvfm_xm_pc <= d_pc;
|
||||
end
|
||||
rvfm_xm_pc_next <=
|
||||
f_jump_now ? f_jump_target :
|
||||
rvfm_x_saw_f_jump ? rvfm_x_saw_f_jump_target :
|
||||
d_pc + (fd_cir_raw[1:0] == 2'b11 ? 32'h4 : 32'h2);
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
reg [31:0] rvfi_pc_rdata_r;
|
||||
reg [31:0] rvfi_pc_wdata_r;
|
||||
|
||||
assign rvfi_pc_rdata = rvfi_pc_rdata_r;
|
||||
assign rvfi_pc_wdata = rvfi_pc_wdata_r;
|
||||
|
||||
always @ (posedge clk) begin
|
||||
if (!m_stall) begin
|
||||
rvfi_pc_rdata_r <= rvfm_xm_pc;
|
||||
rvfi_pc_wdata_r <=
|
||||
m_trap_enter_vld && m_trap_enter_rdy && xm_except != EXCEPT_NONE ?
|
||||
m_trap_addr : rvfm_xm_pc_next;
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Register file monitor:
|
||||
|
||||
// When writeback is suppressed due to trap, the previous instruction is left
|
||||
// in the writeback buffer (and can be re-bypassed from there). Make sure not
|
||||
// to report this as a writeback on RVFI.
|
||||
reg rvfm_writeback_mask;
|
||||
always @ (posedge clk) begin
|
||||
if (!m_stall) begin
|
||||
rvfm_writeback_mask <= m_reg_wen_if_nonzero;
|
||||
end
|
||||
end
|
||||
|
||||
assign rvfi_rd_addr = mw_rd & {5{rvfm_writeback_mask}};
|
||||
assign rvfi_rd_wdata = |mw_rd && rvfm_writeback_mask ? mw_result : 32'h0;
|
||||
|
||||
// Do not reimplement internal bypassing logic. Danger of implementing
|
||||
// it correctly here but incorrectly in core.
|
||||
|
||||
reg [31:0] rvfm_xm_rdata1;
|
||||
reg [31:0] rvfm_xm_rdata2;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
rvfm_xm_rdata1 <= 32'h0;
|
||||
rvfm_xm_rdata2 <= 32'h0;
|
||||
end else if (!x_stall) begin
|
||||
// rs*_bypass may have garbage on them for instructions with *no*
|
||||
// register operands (due to some interesting optimisations), though
|
||||
// d_rs* is still driven to 0 to disable stalling on that register
|
||||
// lane. riscv-formal still likes to see zeroes from x0, so fix that
|
||||
// up here. This shouldn't cover up any bugs, since a
|
||||
// register-operand instruction would still *use* the garbage value.
|
||||
rvfm_xm_rdata1 <= |d_rs1 ? x_rs1_bypass : 32'h0;
|
||||
rvfm_xm_rdata2 <= |d_rs2 ? x_rs2_bypass : 32'h0;
|
||||
end
|
||||
end
|
||||
|
||||
reg [4:0] rvfi_rs1_addr_r;
|
||||
reg [4:0] rvfi_rs2_addr_r;
|
||||
reg [31:0] rvfi_rs1_rdata_r;
|
||||
reg [31:0] rvfi_rs2_rdata_r;
|
||||
|
||||
assign rvfi_rs1_addr = rvfi_rs1_addr_r;
|
||||
assign rvfi_rs2_addr = rvfi_rs2_addr_r;
|
||||
assign rvfi_rs1_rdata = rvfi_rs1_rdata_r;
|
||||
assign rvfi_rs2_rdata = rvfi_rs2_rdata_r;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
rvfi_rs1_addr_r <= 5'h0;
|
||||
rvfi_rs2_addr_r <= 5'h0;
|
||||
rvfi_rs1_rdata_r <= 32'h0;
|
||||
rvfi_rs2_rdata_r <= 32'h0;
|
||||
end else begin
|
||||
rvfi_rs1_addr_r <= m_stall ? 5'h0 : xm_rs1;
|
||||
rvfi_rs2_addr_r <= m_stall ? 5'h0 : xm_rs2;
|
||||
rvfi_rs1_rdata_r <= rvfm_xm_rdata1;
|
||||
rvfi_rs2_rdata_r <= xm_rs2 == mw_rd && |xm_rs2 ? m_wdata : rvfm_xm_rdata2;
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Load/store monitor: based on bus signals, NOT processor internals.
|
||||
// Marshal up a description of the current data phase, and then register this
|
||||
// into the RVFI signals.
|
||||
|
||||
`ifdef HAZARD3_ASSERTIONS
|
||||
`ifndef RISCV_FORMAL_ALIGNED_MEM
|
||||
initial $fatal;
|
||||
`endif
|
||||
`endif
|
||||
|
||||
reg [31:0] rvfm_haddr_dph;
|
||||
reg rvfm_hwrite_dph;
|
||||
reg [1:0] rvfm_htrans_dph;
|
||||
reg [2:0] rvfm_hsize_dph;
|
||||
|
||||
always @ (posedge clk) begin
|
||||
if (bus_aph_ready_d) begin
|
||||
rvfm_htrans_dph <= {bus_aph_req_d, 1'b0};
|
||||
rvfm_haddr_dph <= bus_haddr_d;
|
||||
rvfm_hwrite_dph <= bus_hwrite_d;
|
||||
rvfm_hsize_dph <= bus_hsize_d;
|
||||
end
|
||||
end
|
||||
|
||||
wire [3:0] rvfm_mem_bytemask_dph = (
|
||||
rvfm_hsize_dph == 3'h0 ? 4'h1 :
|
||||
rvfm_hsize_dph == 3'h1 ? 4'h3 :
|
||||
4'hf
|
||||
) << rvfm_haddr_dph[1:0];
|
||||
|
||||
reg [31:0] rvfi_mem_addr_r;
|
||||
reg [3:0] rvfi_mem_rmask_r;
|
||||
reg [31:0] rvfi_mem_rdata_r;
|
||||
reg [3:0] rvfi_mem_wmask_r;
|
||||
reg [31:0] rvfi_mem_wdata_r;
|
||||
reg rvfi_mem_fault_r;
|
||||
// May have to hold the strobes for multiple cycles following a bus
|
||||
// fault, as the trap entry may not go through immediately (depending
|
||||
// on instruction-side bus stall):
|
||||
reg rvfm_mem_hold;
|
||||
|
||||
assign rvfi_mem_addr = rvfi_mem_addr_r;
|
||||
assign rvfi_mem_rdata = rvfi_mem_rdata_r;
|
||||
assign rvfi_mem_wdata = rvfi_mem_wdata_r;
|
||||
|
||||
assign rvfi_mem_fault = rvfi_mem_fault_r;
|
||||
assign rvfi_mem_wmask = rvfi_mem_wmask_r & {4{!rvfi_mem_fault_r}};
|
||||
assign rvfi_mem_rmask = rvfi_mem_rmask_r & {4{!rvfi_mem_fault_r}};
|
||||
assign rvfi_mem_fault_rmask = rvfi_mem_rmask_r & {4{ rvfi_mem_fault_r}};
|
||||
assign rvfi_mem_fault_wmask = rvfi_mem_wmask_r & {4{ rvfi_mem_fault_r}};
|
||||
|
||||
always @ (posedge clk) begin
|
||||
rvfm_mem_hold <= (rvfm_mem_hold || (rvfm_htrans_dph && bus_dph_ready_d)) && m_stall;
|
||||
if (xm_memop == MEMOP_AMO) begin
|
||||
// AMO has completed in stage X. Progressing to stage M without MEMOP
|
||||
// going to NONE then there has been no trap, therefore no stall,
|
||||
// therefore no time for another address to have issued:
|
||||
assert(!m_stall);
|
||||
rvfi_mem_addr_r <= rvfm_haddr_dph;
|
||||
// Always 32-bit, always both read and write:
|
||||
rvfi_mem_rmask_r <= 4'hf;
|
||||
rvfi_mem_wmask_r <= 4'hf;
|
||||
// Has been juggled since the read that matched the winning write:
|
||||
rvfi_mem_rdata_r <= xm_result;
|
||||
// Incidentally captured on previous cycle:
|
||||
rvfi_mem_wdata_r <= rvfi_mem_wdata_r;
|
||||
end else if (bus_dph_ready_d) begin
|
||||
// RVFI has an AXI-like concept of byte strobes, rather than AHB-like
|
||||
rvfi_mem_addr_r <= rvfm_haddr_dph & 32'hffff_fffc;
|
||||
{rvfi_mem_rmask_r, rvfi_mem_wmask_r} <= 0;
|
||||
if (rvfm_htrans_dph[1] && rvfm_hwrite_dph) begin
|
||||
rvfi_mem_wmask_r <= rvfm_mem_bytemask_dph;
|
||||
rvfi_mem_wdata_r <= bus_wdata_d;
|
||||
end else if (rvfm_htrans_dph[1] && !rvfm_hwrite_dph) begin
|
||||
rvfi_mem_rmask_r <= rvfm_mem_bytemask_dph;
|
||||
rvfi_mem_rdata_r <= bus_rdata_d;
|
||||
end
|
||||
rvfi_mem_fault_r <= bus_dph_err_d;
|
||||
end else if (!rvfm_mem_hold) begin
|
||||
rvfi_mem_rmask_r <= 4'h0;
|
||||
rvfi_mem_wmask_r <= 4'h0;
|
||||
rvfi_mem_fault_r <= 1'b0;
|
||||
end
|
||||
// Also need to report rvfi_mem_fault on a fetch fault
|
||||
if (xm_except == EXCEPT_INSTR_FAULT && m_trap_enter_vld && m_trap_enter_rdy) begin
|
||||
rvfi_mem_fault_r <= 1'b1;
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Constraints
|
||||
|
||||
// Trying to keep internal constraints to a minimum.
|
||||
|
||||
// Limit sleep duration for liveness checks
|
||||
// TODO is it possible to do this in a way that doesn't assume the wakeup logic is functional?
|
||||
`ifdef RISCV_FORMAL_FAIRNESS
|
||||
reg [7:0] rvfm_sleep_counter;
|
||||
always @ (posedge clk) begin
|
||||
if (!rst_n) begin
|
||||
rvfm_sleep_counter <= 8'd00;
|
||||
end else if (xm_sleep_wfi || xm_sleep_block) begin
|
||||
rvfm_sleep_counter <= rvfm_sleep_counter + 8'h01;
|
||||
assume(rvfm_sleep_counter < 8'd5);
|
||||
end else begin
|
||||
rvfm_sleep_counter <= 8'd00;
|
||||
end
|
||||
end
|
||||
`endif
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Tie-offs
|
||||
|
||||
// Note: Hazard3 does not have any instructions which irrversibly halt
|
||||
// execution. For the liveness check (RISCV_FORMAL_FAIRNESS is defined),
|
||||
// length of stalls is constrained and WFI is assumed to wake immediately
|
||||
// after going to sleep.
|
||||
assign rvfi_halt = 1'b0;
|
||||
|
||||
// Note: this always reports M-mode, which is not correct if the U_MODE config
|
||||
// is set. However no riscv-formal checks currently use this signal.
|
||||
assign rvfi_mode = 2'h3;
|
||||
|
||||
// Maximum XLEN is always 32 bits
|
||||
assign rvfi_ixl = 2'h1;
|
||||
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2025 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// Macros for building Hazard3 with RVFI trace port outside of the
|
||||
// riscv-formal test harness. For example, when including the trace port in a
|
||||
// synthesised processor.
|
||||
`ifdef HAZARD3_RVFI_STANDALONE
|
||||
|
||||
`define RISCV_FORMAL
|
||||
`define RISCV_FORMAL_NRET 1
|
||||
`define RISCV_FORMAL_XLEN 32
|
||||
`define RISCV_FORMAL_ILEN 32
|
||||
`define RISCV_FORMAL_MEM_FAULT
|
||||
`define RISCV_FORMAL_ALIGNED_MEM
|
||||
|
||||
`define RVFI_OUTPUTS \
|
||||
output wire rvfi_valid, \
|
||||
output wire [63:0] rvfi_order, \
|
||||
output wire [31:0] rvfi_insn, \
|
||||
output wire rvfi_trap, \
|
||||
output wire rvfi_halt, \
|
||||
output wire rvfi_intr, \
|
||||
output wire [1:0] rvfi_mode, \
|
||||
output wire [1:0] rvfi_ixl, \
|
||||
output wire [4:0] rvfi_rs1_addr, \
|
||||
output wire [4:0] rvfi_rs2_addr, \
|
||||
output wire [31:0] rvfi_rs1_rdata, \
|
||||
output wire [31:0] rvfi_rs2_rdata, \
|
||||
output wire [4:0] rvfi_rd_addr, \
|
||||
output wire [31:0] rvfi_rd_wdata, \
|
||||
output wire [31:0] rvfi_pc_rdata, \
|
||||
output wire [31:0] rvfi_pc_wdata, \
|
||||
output wire [31:0] rvfi_mem_addr, \
|
||||
output wire [3:0] rvfi_mem_rmask, \
|
||||
output wire [3:0] rvfi_mem_wmask, \
|
||||
output wire [31:0] rvfi_mem_rdata, \
|
||||
output wire [31:0] rvfi_mem_wdata, \
|
||||
output wire rvfi_mem_fault, \
|
||||
output wire [3:0] rvfi_mem_fault_rmask, \
|
||||
output wire [3:0] rvfi_mem_fault_wmask
|
||||
|
||||
`define RVFI_WIRES \
|
||||
wire rvfi_valid; \
|
||||
wire [63:0] rvfi_order; \
|
||||
wire [31:0] rvfi_insn; \
|
||||
wire rvfi_trap; \
|
||||
wire rvfi_halt; \
|
||||
wire rvfi_intr; \
|
||||
wire [1:0] rvfi_mode; \
|
||||
wire [1:0] rvfi_ixl; \
|
||||
wire [4:0] rvfi_rs1_addr; \
|
||||
wire [4:0] rvfi_rs2_addr; \
|
||||
wire [31:0] rvfi_rs1_rdata; \
|
||||
wire [31:0] rvfi_rs2_rdata; \
|
||||
wire [4:0] rvfi_rd_addr; \
|
||||
wire [31:0] rvfi_rd_wdata; \
|
||||
wire [31:0] rvfi_pc_rdata; \
|
||||
wire [31:0] rvfi_pc_wdata; \
|
||||
wire [31:0] rvfi_mem_addr; \
|
||||
wire [3:0] rvfi_mem_rmask; \
|
||||
wire [3:0] rvfi_mem_wmask; \
|
||||
wire [31:0] rvfi_mem_rdata; \
|
||||
wire [31:0] rvfi_mem_wdata; \
|
||||
wire rvfi_mem_fault; \
|
||||
wire [3:0] rvfi_mem_fault_rmask; \
|
||||
wire [3:0] rvfi_mem_fault_wmask;
|
||||
|
||||
`define RVFI_CONN \
|
||||
.rvfi_valid (rvfi_valid), \
|
||||
.rvfi_order (rvfi_order), \
|
||||
.rvfi_insn (rvfi_insn), \
|
||||
.rvfi_trap (rvfi_trap), \
|
||||
.rvfi_halt (rvfi_halt), \
|
||||
.rvfi_intr (rvfi_intr), \
|
||||
.rvfi_mode (rvfi_mode), \
|
||||
.rvfi_ixl (rvfi_ixl), \
|
||||
.rvfi_rs1_addr (rvfi_rs1_addr), \
|
||||
.rvfi_rs2_addr (rvfi_rs2_addr), \
|
||||
.rvfi_rs1_rdata (rvfi_rs1_rdata), \
|
||||
.rvfi_rs2_rdata (rvfi_rs2_rdata), \
|
||||
.rvfi_rd_addr (rvfi_rd_addr), \
|
||||
.rvfi_rd_wdata (rvfi_rd_wdata), \
|
||||
.rvfi_pc_rdata (rvfi_pc_rdata), \
|
||||
.rvfi_pc_wdata (rvfi_pc_wdata), \
|
||||
.rvfi_mem_addr (rvfi_mem_addr), \
|
||||
.rvfi_mem_rmask (rvfi_mem_rmask), \
|
||||
.rvfi_mem_wmask (rvfi_mem_wmask), \
|
||||
.rvfi_mem_rdata (rvfi_mem_rdata), \
|
||||
.rvfi_mem_wdata (rvfi_mem_wdata), \
|
||||
.rvfi_mem_fault (rvfi_mem_fault), \
|
||||
.rvfi_mem_fault_rmask (rvfi_mem_fault_rmask), \
|
||||
.rvfi_mem_fault_wmask (rvfi_mem_fault_wmask)
|
||||
|
||||
`endif
|
||||
+499
@@ -0,0 +1,499 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
`default_nettype none
|
||||
|
||||
// The Hazard3 trigger unit always implements one trigger of each of the
|
||||
// following types:
|
||||
//
|
||||
// * Instruction count trigger (type=3) with count=1 (can single-step U-mode
|
||||
// from M-mode, or step M-mode foreground from an M-mode exception handler)
|
||||
//
|
||||
// * Interrupt trigger (type=4): trigger on mask of mtip/msip/meip interrupts
|
||||
//
|
||||
// * Exception trigger (type=5): trigger on mask of exception causes
|
||||
//
|
||||
// The following are optionally supported:
|
||||
//
|
||||
// * Instruction address triggers (type=2 execute=1 select=0) aka breakpoints
|
||||
//
|
||||
// Breakpoints always use exact address matches, and the timing is always
|
||||
// "early". The number of breakpoints is configured by BREAKPOINT_TRIGGERS,
|
||||
// which can be 0.
|
||||
//
|
||||
// Interrupt/exception triggers break after the core transfers to its trap
|
||||
// handler, but before the first trap handler instruction executes. Only
|
||||
// action=1 is supported for these triggers, since trap-on-trap is useless
|
||||
// when M is the only privileged mode.
|
||||
|
||||
module hazard3_triggers #(
|
||||
`include "hazard3_config.vh"
|
||||
) (
|
||||
input wire clk,
|
||||
input wire rst_n,
|
||||
|
||||
// Config interface passed through CSR block
|
||||
input wire [11:0] cfg_addr,
|
||||
input wire cfg_wen,
|
||||
input wire [W_DATA-1:0] cfg_wdata,
|
||||
output reg [W_DATA-1:0] cfg_rdata,
|
||||
|
||||
// Global trigger-to-M-mode enable from tcontrol
|
||||
input wire trig_m_en,
|
||||
|
||||
// Fetch address query from stage F
|
||||
input wire [W_ADDR-1:0] fetch_addr,
|
||||
input wire fetch_m_mode,
|
||||
input wire fetch_d_mode,
|
||||
|
||||
// Trap trigger events from stage X
|
||||
input wire event_instr_ret,
|
||||
|
||||
// Trap trigger events from stage M
|
||||
input wire event_interrupt,
|
||||
input wire event_exception,
|
||||
input wire [3:0] event_trap_cause,
|
||||
input wire event_trap_enter,
|
||||
|
||||
// F-aligned break request (for each halfword of word-sized word-aligned fetch)
|
||||
output wire [1:0] break_any,
|
||||
output wire [1:0] break_d_mode,
|
||||
|
||||
// X-aligned step break request (to M-mode only)
|
||||
output wire break_m_step,
|
||||
|
||||
// Stage-X debug mode flag, for CSR protection (may or may not be the same
|
||||
// as the query debug mode flag)
|
||||
input wire x_d_mode,
|
||||
// Stage-X M-mode flag, for enables on interrupt/exception triggers
|
||||
input wire x_m_mode
|
||||
);
|
||||
|
||||
`include "hazard3_csr_addr.vh"
|
||||
|
||||
generate
|
||||
if (DEBUG_SUPPORT == 0) begin: no_triggers
|
||||
|
||||
// The instantiation of this block should already be stubbed out in core.v if
|
||||
// there are no triggers, but we still get warnings for elaborating this
|
||||
// module with zero triggers, so add a generate block here too.
|
||||
|
||||
always @ (*) cfg_rdata = {W_DATA{1'b0}};
|
||||
assign break_any = 1'b0;
|
||||
assign break_d_mode = 1'b0;
|
||||
|
||||
end else begin: have_triggers
|
||||
|
||||
localparam TINDEX_ICOUNT = BREAKPOINT_TRIGGERS + 0;
|
||||
localparam TINDEX_INTERRUPT = BREAKPOINT_TRIGGERS + 1;
|
||||
localparam TINDEX_EXCEPTION = BREAKPOINT_TRIGGERS + 2;
|
||||
localparam N_TRIGGERS = BREAKPOINT_TRIGGERS + 3;
|
||||
|
||||
// If there are no breakpoints, we still have one dummy register (hardwired to
|
||||
// zero) for Verilog wrangling purposes. It has no synthesis effect.
|
||||
localparam N_BREAKPOINT_REGS = BREAKPOINT_TRIGGERS > 0 ? BREAKPOINT_TRIGGERS : 1;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Configuration state
|
||||
|
||||
localparam W_TSELECT = $clog2(N_TRIGGERS);
|
||||
|
||||
reg [W_TSELECT-1:0] tselect;
|
||||
|
||||
// Note tdata1 and mcontrol are the same CSR. tdata1 refers to the universal
|
||||
// fields (type/dmode) and mcontrol refers to those fields specific to
|
||||
// type=2 (address/data match), the only trigger type we implement.
|
||||
|
||||
// State for instruction address match triggers (breakpoints).
|
||||
reg bp_tdata1_dmode [0:N_BREAKPOINT_REGS-1];
|
||||
reg mcontrol_action [0:N_BREAKPOINT_REGS-1];
|
||||
reg mcontrol_m [0:N_BREAKPOINT_REGS-1];
|
||||
reg mcontrol_u [0:N_BREAKPOINT_REGS-1];
|
||||
reg mcontrol_execute [0:N_BREAKPOINT_REGS-1];
|
||||
reg [W_DATA-1:0] bp_tdata2 [0:N_BREAKPOINT_REGS-1];
|
||||
|
||||
// State for instruction count trigger
|
||||
// (hardwired: count=1 dmode=0 action=0; Debug mode single step is already
|
||||
// available via dcsr)
|
||||
reg icount_m;
|
||||
reg icount_u;
|
||||
|
||||
// State for interrupt trigger
|
||||
// (hardwired: action=1; M-mode trap-on-trap is useless as you lose the
|
||||
// original trap state)
|
||||
reg trigger_irq_m;
|
||||
reg trigger_irq_u;
|
||||
reg trigger_irq_dmode;
|
||||
reg [15:0] trigger_irq_cause;
|
||||
|
||||
localparam [15:0] IMPLEMENTED_IRQ_CAUSES = {
|
||||
4'h0, // reserved
|
||||
1'b1, // meip
|
||||
3'h0, // reserved or unimplemented
|
||||
1'b1, // mtip
|
||||
3'h0, // reserved or unimplemented
|
||||
1'b1, // msip
|
||||
3'h0 // reserved or unimplemented
|
||||
};
|
||||
|
||||
// State for exception trigger
|
||||
// (hardwired: action=1; M-mode trap-on-trap is useless as you lose the
|
||||
// original trap state)
|
||||
reg trigger_exception_m;
|
||||
reg trigger_exception_u;
|
||||
reg trigger_exception_dmode;
|
||||
reg [15:0] trigger_exception_cause;
|
||||
|
||||
localparam [15:0] IMPLEMENTED_EXCEPTION_CAUSES = {
|
||||
4'h0, // reserved
|
||||
1'b1, // 11 -> ecall from M-mode
|
||||
2'h0, // reserved or unimplemented
|
||||
|U_MODE, // 8 -> ecall from U-mode
|
||||
1'b1, // 7 -> store/AMO fault
|
||||
1'b1, // 6 -> store/AMO align
|
||||
1'b1, // 5 -> load fault
|
||||
1'b1, // 4 -> load align
|
||||
1'b0, // 3 -> breakpoint; seems useless and risky so disallow
|
||||
1'b1, // 2 -> illegal opcode
|
||||
1'b1, // 1 -> fetch fault
|
||||
~|EXTENSION_C // 0 -> fetch align (only when IALIGN is 32-bit)
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Configuration write port
|
||||
|
||||
localparam N_TRIGGERS_PADDED = 1 << $clog2(N_TRIGGERS);
|
||||
wire [N_TRIGGERS_PADDED-1:0] tselect_match = {{N_TRIGGERS_PADDED-1{1'b0}}, 1'b1} << tselect;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin: cfg_update
|
||||
integer i;
|
||||
if (!rst_n) begin
|
||||
|
||||
tselect <= {W_TSELECT{1'b0}};
|
||||
|
||||
icount_m <= 1'b0;
|
||||
icount_u <= 1'b0;
|
||||
|
||||
trigger_irq_m <= 1'b0;
|
||||
trigger_irq_u <= 1'b0;
|
||||
trigger_irq_dmode <= 1'b0;
|
||||
trigger_irq_cause <= 16'h0;
|
||||
|
||||
trigger_exception_m <= 1'b0;
|
||||
trigger_exception_u <= 1'b0;
|
||||
trigger_exception_dmode <= 1'b0;
|
||||
trigger_exception_cause <= 16'h0;
|
||||
|
||||
for (i = 0; i < BREAKPOINT_TRIGGERS; i = i + 1) begin
|
||||
bp_tdata1_dmode[i] <= 1'b0;
|
||||
mcontrol_action[i] <= 1'b0;
|
||||
mcontrol_m[i] <= 1'b0;
|
||||
mcontrol_u[i] <= 1'b0;
|
||||
mcontrol_execute[i] <= 1'b0;
|
||||
bp_tdata2[i] <= {W_DATA{1'b0}};
|
||||
end
|
||||
|
||||
end else begin
|
||||
|
||||
if (cfg_wen && cfg_addr == TSELECT) begin
|
||||
|
||||
tselect <= cfg_wdata[W_TSELECT-1:0];
|
||||
|
||||
end else if (cfg_wen && cfg_addr == TDATA1) begin
|
||||
|
||||
if (tselect_match[TINDEX_ICOUNT]) begin
|
||||
// This trigger does not implement a dmode bit, as Debug-mode
|
||||
// break on single-step is already provided by dcsr.step
|
||||
icount_m <= cfg_wdata[9];
|
||||
icount_u <= cfg_wdata[6] && |U_MODE;
|
||||
end
|
||||
if (tselect_match[TINDEX_INTERRUPT] && !(trigger_irq_dmode && !x_d_mode)) begin
|
||||
trigger_irq_dmode <= cfg_wdata[27];
|
||||
trigger_irq_m <= cfg_wdata[9];
|
||||
trigger_irq_u <= cfg_wdata[6] && |U_MODE;
|
||||
end
|
||||
if (tselect_match[TINDEX_EXCEPTION] && !(trigger_exception_dmode && !x_d_mode)) begin
|
||||
trigger_exception_dmode <= cfg_wdata[27];
|
||||
trigger_exception_m <= cfg_wdata[9];
|
||||
trigger_exception_u <= cfg_wdata[6] && |U_MODE;
|
||||
end
|
||||
for (i = 0; i < BREAKPOINT_TRIGGERS; i = i + 1) begin
|
||||
if (tselect_match[i] && !(bp_tdata1_dmode[i] && !x_d_mode)) begin
|
||||
if (x_d_mode) begin
|
||||
bp_tdata1_dmode[i] <= cfg_wdata[27];
|
||||
end
|
||||
mcontrol_action[i] <= cfg_wdata[12];
|
||||
mcontrol_m[i] <= cfg_wdata[6];
|
||||
mcontrol_u[i] <= cfg_wdata[3] && |U_MODE;
|
||||
mcontrol_execute[i] <= cfg_wdata[2];
|
||||
end
|
||||
end
|
||||
|
||||
end else if (cfg_wen && cfg_addr == TDATA2) begin
|
||||
|
||||
if (tselect_match[TINDEX_INTERRUPT] && !(trigger_irq_dmode && !x_d_mode)) begin
|
||||
trigger_irq_cause <= cfg_wdata[15:0] & IMPLEMENTED_IRQ_CAUSES;
|
||||
end
|
||||
if (tselect_match[TINDEX_EXCEPTION] && !(trigger_exception_dmode && !x_d_mode)) begin
|
||||
trigger_exception_cause <= cfg_wdata[15:0] & IMPLEMENTED_EXCEPTION_CAUSES;
|
||||
end
|
||||
for (i = 0; i < BREAKPOINT_TRIGGERS; i = i + 1) begin
|
||||
if (tselect_match[i] && !(bp_tdata1_dmode[i] && !x_d_mode)) begin
|
||||
bp_tdata2[i] <= cfg_wdata & {{W_ADDR-2{1'b1}}, |EXTENSION_C, 1'b0};
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
if ((x_d_mode || event_trap_enter) && break_m_step) begin
|
||||
// count field is hardwired, so the trigger is required to disable
|
||||
// itself by clearing its own enables
|
||||
icount_m <= 1'b0;
|
||||
icount_u <= 1'b0;
|
||||
end
|
||||
|
||||
// With no breakpoints, there is still a dummy entry to avoid
|
||||
// `generate` spaghetti; tools complain about comb processes without
|
||||
// sensitivities etc, so just synchronously tie to 0:
|
||||
if (BREAKPOINT_TRIGGERS == 0) begin
|
||||
bp_tdata1_dmode[0] <= 1'b0;
|
||||
mcontrol_action[0] <= 1'b0;
|
||||
mcontrol_m[0] <= 1'b0;
|
||||
mcontrol_u[0] <= 1'b0;
|
||||
mcontrol_execute[0] <= 1'b0;
|
||||
bp_tdata2[0] <= {W_DATA{1'b0}};
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Configuration read port
|
||||
|
||||
reg [W_DATA-1:0] tdata1_rdata [0:N_TRIGGERS_PADDED-1];
|
||||
reg [W_DATA-1:0] tdata2_rdata [0:N_TRIGGERS_PADDED-1];
|
||||
reg [W_DATA-1:0] tinfo_rdata [0:N_TRIGGERS_PADDED-1];
|
||||
|
||||
always @ (*) begin: generate_padded_rdata
|
||||
|
||||
// Default for unimplemented triggers
|
||||
integer i;
|
||||
for (i = 0; i < N_TRIGGERS_PADDED; i = i + 1) begin
|
||||
tdata1_rdata[i] = {W_DATA{1'b0}};
|
||||
tdata2_rdata[i] = {W_DATA{1'b0}};
|
||||
tinfo_rdata[i] = 32'd1 << 0; // type = 0, no trigger
|
||||
end
|
||||
|
||||
// Breakpoints are the first n triggers
|
||||
for (i = 0; i < BREAKPOINT_TRIGGERS; i = i + 1) begin
|
||||
tdata1_rdata[i] = {
|
||||
4'h2, // type = address/data match
|
||||
bp_tdata1_dmode[i],
|
||||
6'h00, // maskmax = 0, exact match only
|
||||
1'b0, // hit = 0, not implemented
|
||||
1'b0, // select = 0, address match only
|
||||
1'b0, // timing = 0, trigger before execution
|
||||
2'h0, // sizelo = 0, unsized
|
||||
{3'h0, mcontrol_action[i]}, // action = 0/1, break to M-mode/D-mode
|
||||
1'b0, // chain = 0, chaining is useless for exact matches
|
||||
4'h0, // match = 0, exact match only
|
||||
mcontrol_m[i],
|
||||
1'b0,
|
||||
1'b0, // s = 0, no S-mode
|
||||
mcontrol_u[i],
|
||||
mcontrol_execute[i],
|
||||
1'b0, // store = 0, this is not a watchpoint
|
||||
1'b0 // load = 0, this is not a watchpoint
|
||||
};
|
||||
tdata2_rdata[i] = bp_tdata2[i];
|
||||
tinfo_rdata[i] = 32'd1 << 2; // type = 2, address/data match
|
||||
end
|
||||
|
||||
// Instruction count trigger
|
||||
tdata1_rdata[TINDEX_ICOUNT] = {
|
||||
4'h3, // type = instruction count
|
||||
1'b0, // dmode = 0 (Debug mode already has dcsr.step)
|
||||
2'h0, // reserved
|
||||
1'b0, // hit = 0
|
||||
14'd1, // count = 1, single-step only
|
||||
icount_m,
|
||||
1'b0, // reserved
|
||||
1'b0, // s = 0, no S-mode
|
||||
icount_u,
|
||||
6'h0 // action = 0, break to M-mode
|
||||
};
|
||||
tinfo_rdata[TINDEX_ICOUNT] = 32'd1 << 3; // type = 3, instruction count
|
||||
|
||||
// Interrupt trigger
|
||||
tdata1_rdata[TINDEX_INTERRUPT] = {
|
||||
4'h4, // type = interrupt
|
||||
trigger_irq_dmode,
|
||||
1'b0, // hit = 0
|
||||
16'h0, // reserved
|
||||
trigger_irq_m,
|
||||
1'b0, // reserved
|
||||
1'b0, // s = 0, no S-mode
|
||||
trigger_irq_u,
|
||||
6'd1 // action = 1, break to Debug mode (if dmode=1)
|
||||
};
|
||||
tdata2_rdata[TINDEX_INTERRUPT] = {
|
||||
16'h0,
|
||||
trigger_irq_cause & IMPLEMENTED_IRQ_CAUSES
|
||||
};
|
||||
tinfo_rdata[TINDEX_INTERRUPT] = 32'd1 << 4;
|
||||
|
||||
// Exception trigger
|
||||
tdata1_rdata[TINDEX_EXCEPTION] = {
|
||||
4'h5, // type = exception
|
||||
trigger_exception_dmode,
|
||||
1'b0, // hit = 0
|
||||
16'h0, // reserved
|
||||
trigger_exception_m,
|
||||
1'b0, // reserved
|
||||
1'b0, // s = 0, no S-mode
|
||||
trigger_exception_u,
|
||||
6'd1 // action = 1, break to Debug mode (if dmode=1)
|
||||
};
|
||||
tdata2_rdata[TINDEX_EXCEPTION] = {
|
||||
16'h0,
|
||||
trigger_exception_cause & IMPLEMENTED_EXCEPTION_CAUSES
|
||||
};
|
||||
tinfo_rdata[TINDEX_EXCEPTION] = 32'd1 << 5;
|
||||
|
||||
end
|
||||
|
||||
always @ (*) begin
|
||||
cfg_rdata = {W_DATA{1'b0}};
|
||||
if (cfg_addr == TSELECT) begin
|
||||
cfg_rdata = {{W_DATA-W_TSELECT{1'b0}}, tselect};
|
||||
end else if (cfg_addr == TDATA1) begin
|
||||
cfg_rdata = tdata1_rdata[tselect];
|
||||
end else if (cfg_addr == TDATA2) begin
|
||||
cfg_rdata = tdata2_rdata[tselect];
|
||||
end else if (cfg_addr == TINFO) begin
|
||||
cfg_rdata = tinfo_rdata[tselect];
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Interrupt/exception trigger logic
|
||||
|
||||
// Ignore tcontrol.mte as these triggers never target M-mode.
|
||||
wire exception_trigger_match =
|
||||
!x_d_mode && trigger_exception_dmode &&
|
||||
(x_m_mode ? trigger_exception_m : trigger_exception_u) &&
|
||||
event_exception &&
|
||||
trigger_exception_cause[event_trap_cause] &&
|
||||
IMPLEMENTED_EXCEPTION_CAUSES[event_trap_cause];
|
||||
|
||||
wire interrupt_trigger_match =
|
||||
!x_d_mode && trigger_irq_dmode &&
|
||||
(x_m_mode ? trigger_irq_m : trigger_irq_u) &&
|
||||
event_interrupt &&
|
||||
trigger_irq_cause[event_trap_cause] &&
|
||||
IMPLEMENTED_IRQ_CAUSES[event_trap_cause];
|
||||
|
||||
// Asserted no later than the end of the aphase for the instruction fetch at
|
||||
// mtvec. Tags the dphase of trap handler instruction fetches as containing
|
||||
// breakpoints.
|
||||
reg break_ie;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
break_ie <= 1'b0;
|
||||
end else begin
|
||||
break_ie <= !x_d_mode && (break_ie || (
|
||||
exception_trigger_match || interrupt_trigger_match
|
||||
));
|
||||
end
|
||||
end
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Instruction count trigger logic (single-step under M-mode control)
|
||||
|
||||
wire step_break_enabled = trig_m_en && !x_d_mode && (
|
||||
x_m_mode ? icount_m : icount_u
|
||||
);
|
||||
|
||||
reg break_on_step;
|
||||
|
||||
always @ (posedge clk or negedge rst_n) begin
|
||||
if (!rst_n) begin
|
||||
break_on_step <= 1'b0;
|
||||
end else begin
|
||||
// Note icount triggers differ from dcsr.step in that they ignore
|
||||
// exceptions, only triggering on retired instructions.
|
||||
break_on_step <= !(x_d_mode || event_trap_enter) && (break_on_step || (
|
||||
event_instr_ret && step_break_enabled
|
||||
));
|
||||
end
|
||||
end
|
||||
|
||||
assign break_m_step = break_on_step;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Breakpoint trigger logic
|
||||
|
||||
// To reduce the fanin of jump and load/store gating in stage X, the address
|
||||
// lookup is in stage F (fetch data phase). We check *fetch addresses*, not
|
||||
// program counter values. Fetches are always word-sized and word-aligned.
|
||||
//
|
||||
// To ensure it is safe to do this, non-debug-mode writes to the TDATA1 and
|
||||
// TDATA2 CSRs cause a prefetch flush, to maintain write-to-fetch ordering.
|
||||
//
|
||||
// It's possible for different breakpoints to match different halfwords of the
|
||||
// fetch word. The trigger unit must report both matches separately, because
|
||||
// it is not known at this point where the instruction boundaries are (we
|
||||
// don't have the instruction data yet).
|
||||
|
||||
wire [N_BREAKPOINT_REGS-1:0] breakpoint_enabled;
|
||||
wire [N_BREAKPOINT_REGS-1:0] breakpoint_match;
|
||||
wire [N_BREAKPOINT_REGS-1:0] want_d_mode_break;
|
||||
wire [N_BREAKPOINT_REGS-1:0] want_m_mode_break;
|
||||
wire [N_BREAKPOINT_REGS-1:0] want_d_mode_break_hw0;
|
||||
wire [N_BREAKPOINT_REGS-1:0] want_d_mode_break_hw1;
|
||||
wire [N_BREAKPOINT_REGS-1:0] want_m_mode_break_hw0;
|
||||
wire [N_BREAKPOINT_REGS-1:0] want_m_mode_break_hw1;
|
||||
|
||||
genvar g;
|
||||
for (g = 0; g < N_BREAKPOINT_REGS; g = g + 1) begin: match_pc
|
||||
// Detect tripped breakpoints
|
||||
assign breakpoint_enabled[g] = mcontrol_execute[g] && !fetch_d_mode && (
|
||||
fetch_m_mode ? mcontrol_m[g] : mcontrol_u[g]
|
||||
);
|
||||
assign breakpoint_match[g] = breakpoint_enabled[g] && fetch_addr == {bp_tdata2[g][W_DATA-1:2], 2'b00};
|
||||
// Decide the type of break implied by the trip
|
||||
assign want_d_mode_break[g] = breakpoint_match[g] && mcontrol_action[g] && bp_tdata1_dmode[g];
|
||||
assign want_m_mode_break[g] = breakpoint_match[g] && !mcontrol_action[g] && trig_m_en;
|
||||
// Report separately for each halfword, so the frontend can pass this
|
||||
// through the prefetch buffer. A breakpoint exception is taken when
|
||||
// the first halfword of an instruction (of any size) is flagged with
|
||||
// a breakpoint, implying an exact match.
|
||||
assign want_d_mode_break_hw0[g] = want_d_mode_break[g] && !bp_tdata2[g][1];
|
||||
assign want_d_mode_break_hw1[g] = want_d_mode_break[g] && bp_tdata2[g][1];
|
||||
assign want_m_mode_break_hw0[g] = want_m_mode_break[g] && !bp_tdata2[g][1];
|
||||
assign want_m_mode_break_hw1[g] = want_m_mode_break[g] && bp_tdata2[g][1];
|
||||
end
|
||||
|
||||
// Break flags to frontend (tag the current fetch dphase as containing a breakpoint):
|
||||
|
||||
assign break_any = {
|
||||
|want_m_mode_break_hw1 || |want_d_mode_break_hw1 || break_ie,
|
||||
|want_m_mode_break_hw0 || |want_d_mode_break_hw0 || break_ie
|
||||
} & {2{BREAKPOINT_TRIGGERS > 0}};
|
||||
|
||||
assign break_d_mode = {
|
||||
|want_d_mode_break_hw1 || break_ie,
|
||||
|want_d_mode_break_hw0 || break_ie
|
||||
} & {2{BREAKPOINT_TRIGGERS > 0}};
|
||||
|
||||
end
|
||||
endgenerate
|
||||
|
||||
endmodule
|
||||
|
||||
`ifndef YOSYS
|
||||
`default_nettype wire
|
||||
`endif
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
// These really ought to be localparams, but are occasionally needed for
|
||||
// passing flags around between modules, so are made available as parameters
|
||||
// instead. It's ugly, but better scope hygiene than the preprocessor. These
|
||||
// parameters should not be changed from their default values.
|
||||
|
||||
parameter W_REGADDR = 5,
|
||||
|
||||
parameter W_ALUOP = 6,
|
||||
parameter W_ALUSRC = 1,
|
||||
parameter W_MEMOP = 5,
|
||||
parameter W_BCOND = 2,
|
||||
parameter W_SHAMT = 5,
|
||||
|
||||
parameter W_EXCEPT = 4,
|
||||
parameter W_MULOP = 3
|
||||
Vendored
+276
@@ -0,0 +1,276 @@
|
||||
/*****************************************************************************\
|
||||
| Copyright (C) 2021-2022 Luke Wren |
|
||||
| SPDX-License-Identifier: Apache-2.0 |
|
||||
\*****************************************************************************/
|
||||
|
||||
localparam RV_RS1_LSB = 15;
|
||||
localparam RV_RS1_BITS = 5;
|
||||
localparam RV_RS2_LSB = 20;
|
||||
localparam RV_RS2_BITS = 5;
|
||||
localparam RV_RD_LSB = 7;
|
||||
localparam RV_RD_BITS = 5;
|
||||
|
||||
// Note: these are preprocessor macros, rather than the usual localparams,
|
||||
// because it's quite difficult to get a definitive citation from 1364-2005
|
||||
// for whether Z values are propagated through a localparam to a casez.
|
||||
// Multiple tools complain about it, so just this once I'll use macros.
|
||||
|
||||
`ifndef HAZARD3_RVOPC_MACROS
|
||||
`define HAZARD3_RVOPC_MACROS
|
||||
|
||||
// Base ISA (some of these are Z now)
|
||||
`define RVOPC_BEQ 32'b?????????????????000?????1100011
|
||||
`define RVOPC_BNE 32'b?????????????????001?????1100011
|
||||
`define RVOPC_BLT 32'b?????????????????100?????1100011
|
||||
`define RVOPC_BGE 32'b?????????????????101?????1100011
|
||||
`define RVOPC_BLTU 32'b?????????????????110?????1100011
|
||||
`define RVOPC_BGEU 32'b?????????????????111?????1100011
|
||||
`define RVOPC_JALR 32'b?????????????????000?????1100111
|
||||
`define RVOPC_JAL 32'b?????????????????????????1101111
|
||||
`define RVOPC_LUI 32'b?????????????????????????0110111
|
||||
`define RVOPC_AUIPC 32'b?????????????????????????0010111
|
||||
`define RVOPC_ADDI 32'b?????????????????000?????0010011
|
||||
`define RVOPC_SLLI 32'b0000000??????????001?????0010011
|
||||
`define RVOPC_SLTI 32'b?????????????????010?????0010011
|
||||
`define RVOPC_SLTIU 32'b?????????????????011?????0010011
|
||||
`define RVOPC_XORI 32'b?????????????????100?????0010011
|
||||
`define RVOPC_SRLI 32'b0000000??????????101?????0010011
|
||||
`define RVOPC_SRAI 32'b0100000??????????101?????0010011
|
||||
`define RVOPC_ORI 32'b?????????????????110?????0010011
|
||||
`define RVOPC_ANDI 32'b?????????????????111?????0010011
|
||||
`define RVOPC_ADD 32'b0000000??????????000?????0110011
|
||||
`define RVOPC_SUB 32'b0100000??????????000?????0110011
|
||||
`define RVOPC_SLL 32'b0000000??????????001?????0110011
|
||||
`define RVOPC_SLT 32'b0000000??????????010?????0110011
|
||||
`define RVOPC_SLTU 32'b0000000??????????011?????0110011
|
||||
`define RVOPC_XOR 32'b0000000??????????100?????0110011
|
||||
`define RVOPC_SRL 32'b0000000??????????101?????0110011
|
||||
`define RVOPC_SRA 32'b0100000??????????101?????0110011
|
||||
`define RVOPC_OR 32'b0000000??????????110?????0110011
|
||||
`define RVOPC_AND 32'b0000000??????????111?????0110011
|
||||
`define RVOPC_LB 32'b?????????????????000?????0000011
|
||||
`define RVOPC_LH 32'b?????????????????001?????0000011
|
||||
`define RVOPC_LW 32'b?????????????????010?????0000011
|
||||
`define RVOPC_LBU 32'b?????????????????100?????0000011
|
||||
`define RVOPC_LHU 32'b?????????????????101?????0000011
|
||||
`define RVOPC_SB 32'b?????????????????000?????0100011
|
||||
`define RVOPC_SH 32'b?????????????????001?????0100011
|
||||
`define RVOPC_SW 32'b?????????????????010?????0100011
|
||||
`define RVOPC_FENCE 32'b????????????00000000000000001111
|
||||
`define RVOPC_FENCE_I 32'b00000000000000000001000000001111
|
||||
`define RVOPC_ECALL 32'b00000000000000000000000001110011
|
||||
`define RVOPC_EBREAK 32'b00000000000100000000000001110011
|
||||
`define RVOPC_CSRRW 32'b?????????????????001?????1110011
|
||||
`define RVOPC_CSRRS 32'b?????????????????010?????1110011
|
||||
`define RVOPC_CSRRC 32'b?????????????????011?????1110011
|
||||
`define RVOPC_CSRRWI 32'b?????????????????101?????1110011
|
||||
`define RVOPC_CSRRSI 32'b?????????????????110?????1110011
|
||||
`define RVOPC_CSRRCI 32'b?????????????????111?????1110011
|
||||
`define RVOPC_MRET 32'b00110000001000000000000001110011
|
||||
`define RVOPC_SYSTEM 32'b?????????????????????????1110011
|
||||
`define RVOPC_WFI 32'b00010000010100000000000001110011
|
||||
|
||||
// M extension
|
||||
`define RVOPC_MUL 32'b0000001??????????000?????0110011
|
||||
`define RVOPC_MULH 32'b0000001??????????001?????0110011
|
||||
`define RVOPC_MULHSU 32'b0000001??????????010?????0110011
|
||||
`define RVOPC_MULHU 32'b0000001??????????011?????0110011
|
||||
`define RVOPC_DIV 32'b0000001??????????100?????0110011
|
||||
`define RVOPC_DIVU 32'b0000001??????????101?????0110011
|
||||
`define RVOPC_REM 32'b0000001??????????110?????0110011
|
||||
`define RVOPC_REMU 32'b0000001??????????111?????0110011
|
||||
|
||||
// A extension
|
||||
`define RVOPC_LR_W 32'b00010??00000?????010?????0101111
|
||||
`define RVOPC_SC_W 32'b00011????????????010?????0101111
|
||||
`define RVOPC_AMOSWAP_W 32'b00001????????????010?????0101111
|
||||
`define RVOPC_AMOADD_W 32'b00000????????????010?????0101111
|
||||
`define RVOPC_AMOXOR_W 32'b00100????????????010?????0101111
|
||||
`define RVOPC_AMOAND_W 32'b01100????????????010?????0101111
|
||||
`define RVOPC_AMOOR_W 32'b01000????????????010?????0101111
|
||||
`define RVOPC_AMOMIN_W 32'b10000????????????010?????0101111
|
||||
`define RVOPC_AMOMAX_W 32'b10100????????????010?????0101111
|
||||
`define RVOPC_AMOMINU_W 32'b11000????????????010?????0101111
|
||||
`define RVOPC_AMOMAXU_W 32'b11100????????????010?????0101111
|
||||
|
||||
// Zba (address generation)
|
||||
`define RVOPC_SH1ADD 32'b0010000??????????010?????0110011
|
||||
`define RVOPC_SH2ADD 32'b0010000??????????100?????0110011
|
||||
`define RVOPC_SH3ADD 32'b0010000??????????110?????0110011
|
||||
|
||||
// Zbb (basic bit manipulation)
|
||||
`define RVOPC_ANDN 32'b0100000??????????111?????0110011
|
||||
`define RVOPC_CLZ 32'b011000000000?????001?????0010011
|
||||
`define RVOPC_CPOP 32'b011000000010?????001?????0010011
|
||||
`define RVOPC_CTZ 32'b011000000001?????001?????0010011
|
||||
`define RVOPC_MAX 32'b0000101??????????110?????0110011
|
||||
`define RVOPC_MAXU 32'b0000101??????????111?????0110011
|
||||
`define RVOPC_MIN 32'b0000101??????????100?????0110011
|
||||
`define RVOPC_MINU 32'b0000101??????????101?????0110011
|
||||
`define RVOPC_ORC_B 32'b001010000111?????101?????0010011
|
||||
`define RVOPC_ORN 32'b0100000??????????110?????0110011
|
||||
`define RVOPC_REV8 32'b011010011000?????101?????0010011
|
||||
`define RVOPC_ROL 32'b0110000??????????001?????0110011
|
||||
`define RVOPC_ROR 32'b0110000??????????101?????0110011
|
||||
`define RVOPC_RORI 32'b0110000??????????101?????0010011
|
||||
`define RVOPC_SEXT_B 32'b011000000100?????001?????0010011
|
||||
`define RVOPC_SEXT_H 32'b011000000101?????001?????0010011
|
||||
`define RVOPC_XNOR 32'b0100000??????????100?????0110011
|
||||
`define RVOPC_ZEXT_H 32'b000010000000?????100?????0110011
|
||||
|
||||
// Zbc (carry-less multiply)
|
||||
`define RVOPC_CLMUL 32'b0000101??????????001?????0110011
|
||||
`define RVOPC_CLMULH 32'b0000101??????????011?????0110011
|
||||
`define RVOPC_CLMULR 32'b0000101??????????010?????0110011
|
||||
|
||||
// Zbs (single-bit manipulation)
|
||||
`define RVOPC_BCLR 32'b0100100??????????001?????0110011
|
||||
`define RVOPC_BCLRI 32'b0100100??????????001?????0010011
|
||||
`define RVOPC_BEXT 32'b0100100??????????101?????0110011
|
||||
`define RVOPC_BEXTI 32'b0100100??????????101?????0010011
|
||||
`define RVOPC_BINV 32'b0110100??????????001?????0110011
|
||||
`define RVOPC_BINVI 32'b0110100??????????001?????0010011
|
||||
`define RVOPC_BSET 32'b0010100??????????001?????0110011
|
||||
`define RVOPC_BSETI 32'b0010100??????????001?????0010011
|
||||
|
||||
// Zbkb (basic bit manipulation for crypto) (minus those in Zbb)
|
||||
`define RVOPC_PACK 32'b0000100??????????100?????0110011
|
||||
`define RVOPC_PACKH 32'b0000100??????????111?????0110011
|
||||
`define RVOPC_BREV8 32'b011010000111?????101?????0010011
|
||||
`define RVOPC_UNZIP 32'b000010001111?????101?????0010011
|
||||
`define RVOPC_ZIP 32'b000010001111?????001?????0010011
|
||||
|
||||
// Zbkc is a subset of Zbc.
|
||||
|
||||
// Zbkx (crossbar permutation)
|
||||
`define RVOPC_XPERM8 32'b0010100??????????100?????0110011
|
||||
`define RVOPC_XPERM4 32'b0010100??????????010?????0110011
|
||||
|
||||
// Zilsd (load/store pair)
|
||||
`define RVOPC_LD 32'b?????????????????011????00000011 // rd[0] == 0
|
||||
`define RVOPC_SD 32'b???????????0?????011?????0100011 // rs2[0] == 0
|
||||
|
||||
// Hazard3 custom instructions
|
||||
|
||||
// Xh3bextm (Hazard3 multi-bit extract): multi-bit versions of bext/bexti from Zbs
|
||||
`define RVOPC_H3_BEXTM 32'b000???0??????????000?????0001011 // custom-0 funct3=0
|
||||
`define RVOPC_H3_BEXTMI 32'b000???0??????????100?????0001011 // custom-0 funct3=4
|
||||
|
||||
// C Extension
|
||||
`define RVOPC_C_ADDI4SPN 16'b000???????????00 // *** illegal if imm 0
|
||||
`define RVOPC_C_LW 16'b010???????????00
|
||||
`define RVOPC_C_SW 16'b110???????????00
|
||||
|
||||
`define RVOPC_C_ADDI 16'b000???????????01
|
||||
`define RVOPC_C_JAL 16'b001???????????01
|
||||
`define RVOPC_C_J 16'b101???????????01
|
||||
`define RVOPC_C_LI 16'b010???????????01
|
||||
// addi16sp when rd=2:
|
||||
`define RVOPC_C_LUI 16'b011???????????01 // *** reserved if imm 0 (for both LUI and ADDI16SP)
|
||||
`define RVOPC_C_SRLI 16'b100000????????01 // On RV32 imm[5] (instr[12]) must be 0, else reserved NSE.
|
||||
`define RVOPC_C_SRAI 16'b100001????????01 // On RV32 imm[5] (instr[12]) must be 0, else reserved NSE.
|
||||
`define RVOPC_C_ANDI 16'b100?10????????01
|
||||
`define RVOPC_C_SUB 16'b100011???00???01
|
||||
`define RVOPC_C_XOR 16'b100011???01???01
|
||||
`define RVOPC_C_OR 16'b100011???10???01
|
||||
`define RVOPC_C_AND 16'b100011???11???01
|
||||
`define RVOPC_C_BEQZ 16'b110???????????01
|
||||
`define RVOPC_C_BNEZ 16'b111???????????01
|
||||
|
||||
`define RVOPC_C_SLLI 16'b0000??????????10 // On RV32 imm[5] (instr[12]) must be 0, else reserved NSE.
|
||||
// jr if !rs2:
|
||||
`define RVOPC_C_MV 16'b1000??????????10 // *** reserved if JR and !rs1 (instr[11:7])
|
||||
// jalr if !rs2:
|
||||
`define RVOPC_C_ADD 16'b1001??????????10 // *** EBREAK if !instr[11:2]
|
||||
`define RVOPC_C_LWSP 16'b010???????????10 // *** reserved if rd=x0
|
||||
`define RVOPC_C_SWSP 16'b110???????????10
|
||||
|
||||
// Zcb simple additional compressed instructions
|
||||
`define RVOPC_C_LBU 16'b100000????????00
|
||||
`define RVOPC_C_LHU 16'b100001???0????00
|
||||
`define RVOPC_C_LH 16'b100001???1????00
|
||||
`define RVOPC_C_SB 16'b100010????????00
|
||||
`define RVOPC_C_SH 16'b100011???0????00
|
||||
`define RVOPC_C_ZEXT_B 16'b100111???1100001
|
||||
`define RVOPC_C_SEXT_B 16'b100111???1100101
|
||||
`define RVOPC_C_ZEXT_H 16'b100111???1101001
|
||||
`define RVOPC_C_SEXT_H 16'b100111???1101101
|
||||
`define RVOPC_C_NOT 16'b100111???1110101
|
||||
`define RVOPC_C_MUL 16'b100111???10???01
|
||||
|
||||
// Zclsd load/store pair instructions
|
||||
`define RVOPC_C_LD 16'b011??????????000 // rd[0] == 0
|
||||
`define RVOPC_C_LDSP 16'b011?????0?????10 // rd[0] == 0
|
||||
`define RVOPC_C_SD 16'b111??????????000 // rs2[0] == 0
|
||||
`define RVOPC_C_SDSP 16'b111??????????010 // rs2[0] == 0
|
||||
|
||||
// Zcmp push/pop instructions
|
||||
`define RVOPC_CM_PUSH 16'b10111000??????10
|
||||
`define RVOPC_CM_POP 16'b10111010??????10
|
||||
`define RVOPC_CM_POPRETZ 16'b10111100??????10
|
||||
`define RVOPC_CM_POPRET 16'b10111110??????10
|
||||
`define RVOPC_CM_MVSA01 16'b101011???01???10
|
||||
`define RVOPC_CM_MVA01S 16'b101011???11???10
|
||||
|
||||
// Copies provided here with 0 instead of ? so that these can be used to build 32-bit instructions in the decompressor
|
||||
|
||||
`define RVOPC_NOZ_BEQ 32'b00000000000000000000000001100011
|
||||
`define RVOPC_NOZ_BNE 32'b00000000000000000001000001100011
|
||||
`define RVOPC_NOZ_BLT 32'b00000000000000000100000001100011
|
||||
`define RVOPC_NOZ_BGE 32'b00000000000000000101000001100011
|
||||
`define RVOPC_NOZ_BLTU 32'b00000000000000000110000001100011
|
||||
`define RVOPC_NOZ_BGEU 32'b00000000000000000111000001100011
|
||||
`define RVOPC_NOZ_JALR 32'b00000000000000000000000001100111
|
||||
`define RVOPC_NOZ_JAL 32'b00000000000000000000000001101111
|
||||
`define RVOPC_NOZ_LUI 32'b00000000000000000000000000110111
|
||||
`define RVOPC_NOZ_AUIPC 32'b00000000000000000000000000010111
|
||||
`define RVOPC_NOZ_ADDI 32'b00000000000000000000000000010011
|
||||
`define RVOPC_NOZ_SLLI 32'b00000000000000000001000000010011
|
||||
`define RVOPC_NOZ_SLTI 32'b00000000000000000010000000010011
|
||||
`define RVOPC_NOZ_SLTIU 32'b00000000000000000011000000010011
|
||||
`define RVOPC_NOZ_XORI 32'b00000000000000000100000000010011
|
||||
`define RVOPC_NOZ_SRLI 32'b00000000000000000101000000010011
|
||||
`define RVOPC_NOZ_SRAI 32'b01000000000000000101000000010011
|
||||
`define RVOPC_NOZ_ORI 32'b00000000000000000110000000010011
|
||||
`define RVOPC_NOZ_ANDI 32'b00000000000000000111000000010011
|
||||
`define RVOPC_NOZ_ADD 32'b00000000000000000000000000110011
|
||||
`define RVOPC_NOZ_SUB 32'b01000000000000000000000000110011
|
||||
`define RVOPC_NOZ_SLL 32'b00000000000000000001000000110011
|
||||
`define RVOPC_NOZ_SLT 32'b00000000000000000010000000110011
|
||||
`define RVOPC_NOZ_SLTU 32'b00000000000000000011000000110011
|
||||
`define RVOPC_NOZ_XOR 32'b00000000000000000100000000110011
|
||||
`define RVOPC_NOZ_SRL 32'b00000000000000000101000000110011
|
||||
`define RVOPC_NOZ_SRA 32'b01000000000000000101000000110011
|
||||
`define RVOPC_NOZ_OR 32'b00000000000000000110000000110011
|
||||
`define RVOPC_NOZ_AND 32'b00000000000000000111000000110011
|
||||
`define RVOPC_NOZ_LB 32'b00000000000000000000000000000011
|
||||
`define RVOPC_NOZ_LH 32'b00000000000000000001000000000011
|
||||
`define RVOPC_NOZ_LW 32'b00000000000000000010000000000011
|
||||
`define RVOPC_NOZ_LBU 32'b00000000000000000100000000000011
|
||||
`define RVOPC_NOZ_LHU 32'b00000000000000000101000000000011
|
||||
`define RVOPC_NOZ_SB 32'b00000000000000000000000000100011
|
||||
`define RVOPC_NOZ_SH 32'b00000000000000000001000000100011
|
||||
`define RVOPC_NOZ_SW 32'b00000000000000000010000000100011
|
||||
`define RVOPC_NOZ_FENCE 32'b00000000000000000000000000001111
|
||||
`define RVOPC_NOZ_FENCE_I 32'b00000000000000000001000000001111
|
||||
`define RVOPC_NOZ_ECALL 32'b00000000000000000000000001110011
|
||||
`define RVOPC_NOZ_EBREAK 32'b00000000000100000000000001110011
|
||||
`define RVOPC_NOZ_CSRRW 32'b00000000000000000001000001110011
|
||||
`define RVOPC_NOZ_CSRRS 32'b00000000000000000010000001110011
|
||||
`define RVOPC_NOZ_CSRRC 32'b00000000000000000011000001110011
|
||||
`define RVOPC_NOZ_CSRRWI 32'b00000000000000000101000001110011
|
||||
`define RVOPC_NOZ_CSRRSI 32'b00000000000000000110000001110011
|
||||
`define RVOPC_NOZ_CSRRCI 32'b00000000000000000111000001110011
|
||||
`define RVOPC_NOZ_SYSTEM 32'b00000000000000000000000001110011
|
||||
|
||||
// Non-RV32I instructions for Zcb:
|
||||
`define RVOPC_NOZ_MUL 32'b00000010000000000000000000110011
|
||||
`define RVOPC_NOZ_SEXT_B 32'b01100000010000000001000000010011
|
||||
`define RVOPC_NOZ_SEXT_H 32'b01100000010100000001000000010011
|
||||
`define RVOPC_NOZ_ZEXT_H 32'b00001000000000000100000000110011
|
||||
|
||||
// Non-RV32I instructions for Zclsd:
|
||||
`define RVOPC_NOZ_LD 32'b00000000000000000011000000000011
|
||||
`define RVOPC_NOZ_SD 32'b00000000000000000011000000100011
|
||||
|
||||
`endif
|
||||
Reference in New Issue
Block a user