diff --git a/hdl/ip/vhd/axi_blocks/BUCK b/hdl/ip/vhd/axi_blocks/BUCK
index 66ede9a4..a939e1e0 100644
--- a/hdl/ip/vhd/axi_blocks/BUCK
+++ b/hdl/ip/vhd/axi_blocks/BUCK
@@ -1,4 +1,4 @@
-load("//tools:hdl.bzl", "vhdl_unit")
+load("//tools:hdl.bzl", "vhdl_unit", "vunit_sim")
vhdl_unit(
name = "axilite_common_pkg",
@@ -86,11 +86,32 @@ vhdl_unit(
visibility = ['PUBLIC'],
)
+vhdl_unit(
+ name = "axil_pipe",
+ srcs = glob(["axil_pipe.vhd"]),
+ deps = [
+ ":axilite_common_pkg",
+ ],
+ standard = "2008",
+ visibility = ['PUBLIC'],
+)
+
vhdl_unit(
name = "axil_interconnect_2k8",
srcs = glob(["axil_interconnect_2k8.vhd"]),
deps = [
":axilite_if_2k8",
+ ":axil_pipe",
+ ],
+ standard = "2008",
+ visibility = ['PUBLIC'],
+)
+
+vunit_sim(
+ name = "axil_interconnect_tb",
+ srcs = glob(["sims/*.vhd"]),
+ deps = [
+ ":axil_interconnect_2k8",
],
standard = "2008",
visibility = ['PUBLIC'],
diff --git a/hdl/ip/vhd/axi_blocks/axil8_resizer.vhd b/hdl/ip/vhd/axi_blocks/axil8_resizer.vhd
index 6e51eb0d..04c07165 100644
--- a/hdl/ip/vhd/axi_blocks/axil8_resizer.vhd
+++ b/hdl/ip/vhd/axi_blocks/axil8_resizer.vhd
@@ -31,13 +31,15 @@ begin
responder.write_data.strb <= fabric.write_data.strb;
responder.write_response.ready <= fabric.write_response.ready;
- fabric.read_address.valid <= responder.read_address.valid;
- responder.read_address.addr <= fabric.read_address.addr(responder.read_address.addr'length - 1 downto 0);
- fabric.read_data.ready <= responder.read_data.ready;
fabric.write_response.resp <= responder.write_response.resp;
fabric.write_response.valid <= responder.write_response.valid;
+
+ responder.read_address.valid <= fabric.read_address.valid;
+ responder.read_address.addr <= fabric.read_address.addr(responder.read_address.addr'length - 1 downto 0);
fabric.read_address.ready <= responder.read_address.ready;
- responder.read_data.resp <= fabric.read_data.resp;
- responder.read_data.valid <= fabric.read_data.valid;
+
+ responder.read_data.ready <= fabric.read_data.ready;
+ fabric.read_data.resp <= responder.read_data.resp;
+ fabric.read_data.valid <= responder.read_data.valid;
fabric.read_data.data <= responder.read_data.data;
end rtl;
\ No newline at end of file
diff --git a/hdl/ip/vhd/axi_blocks/axil_common_pkg.vhd b/hdl/ip/vhd/axi_blocks/axil_common_pkg.vhd
index 76e59ed7..b10687de 100644
--- a/hdl/ip/vhd/axi_blocks/axil_common_pkg.vhd
+++ b/hdl/ip/vhd/axi_blocks/axil_common_pkg.vhd
@@ -13,14 +13,133 @@ package axil_common_pkg is
type axil_responder_config is record
base_addr : std_logic_vector(31 downto 0);
addr_span_bits : integer;
+ -- Register stages the interconnect inserts in *each* direction between
+ -- the fabric and this responder, to buy back setup time on a responder
+ -- that is physically far from the fabric. Added round trip latency is
+ -- 2 * pipe_stages cycles plus a few fixed handshake cycles, so raise it
+ -- only where timing needs it. 0 is a zero cost pass-through.
+ pipe_stages : natural;
end record;
type axil_responder_cfg_array_t is array (natural range <>) of axil_responder_config;
+ -- Constructor for a responder config, so adding a field later does not
+ -- break every aggregate in the tree.
+ function resp_cfg (
+ constant base_addr : std_logic_vector(31 downto 0);
+ constant addr_span_bits : integer;
+ constant pipe_stages : natural := 0
+ ) return axil_responder_config;
+
type int_array is array (natural range <>) of integer;
constant OKAY : std_logic_vector(1 downto 0) := "00";
constant EXOKAY : std_logic_vector(1 downto 0) := "01";
constant SLVERR : std_logic_vector(1 downto 0) := "10";
+ -- Ones in the low addr_span_bits positions, zeros above. Used both to mask
+ -- a fabric address down to what a responder actually decodes and to select
+ -- the bits that participate in the address compare.
+ function span_mask (constant addr_span_bits : integer) return std_logic_vector;
+
+ -- Every responder base address must be aligned to its own span, otherwise
+ -- the upper-bits equality compare in the interconnect is not equivalent to
+ -- a base/limit compare.
+ function bases_aligned (constant cfg : axil_responder_cfg_array_t) return boolean;
+
+ -- No two responder address ranges may overlap, otherwise more than one bit
+ -- of the interconnect's one-hot select can be set at once.
+ function ranges_disjoint (constant cfg : axil_responder_cfg_array_t) return boolean;
+
+ -- Every responder must be reachable from an initiator of the given address
+ -- width, otherwise it silently falls through to the error responder.
+ function bases_reachable (
+ constant cfg : axil_responder_cfg_array_t;
+ constant initiator_addr_width : integer
+ ) return boolean;
+
end package;
+
+package body axil_common_pkg is
+
+ function resp_cfg (
+ constant base_addr : std_logic_vector(31 downto 0);
+ constant addr_span_bits : integer;
+ constant pipe_stages : natural := 0
+ ) return axil_responder_config is
+ begin
+ return (base_addr => base_addr,
+ addr_span_bits => addr_span_bits,
+ pipe_stages => pipe_stages);
+ end function;
+
+ function span_mask (constant addr_span_bits : integer) return std_logic_vector is
+ variable mask : std_logic_vector(31 downto 0) := (others => '0');
+ begin
+ -- Set every bit below addr_span_bits. Walking the bits one at a time
+ -- rather than slicing keeps addr_span_bits out of a slice bound, so this
+ -- stays legal for the 0 and 32 cases (which would be null slices) and
+ -- stays synthesizable under both Vivado and ghdl.
+ for i in mask'reverse_range loop
+ if i < addr_span_bits then
+ mask(i) := '1';
+ end if;
+ end loop;
+ return mask;
+ end function;
+
+ function bases_aligned (constant cfg : axil_responder_cfg_array_t) return boolean is
+ begin
+ -- Visit every responder and bail out on the first misaligned base. A base
+ -- is aligned when none of the bits inside its own span are set, which is
+ -- exactly the bits span_mask selects.
+ for i in cfg'range loop
+ if (cfg(i).base_addr and span_mask(cfg(i).addr_span_bits)) /= 32x"0" then
+ return false;
+ end if;
+ end loop;
+ return true;
+ end function;
+
+ function ranges_disjoint (constant cfg : axil_responder_cfg_array_t) return boolean is
+ variable wider : integer;
+ begin
+ -- Visit each unordered pair of responders once (the j > i guard is what
+ -- skips the self comparison and the mirror of a pair already checked) and
+ -- bail out on the first overlap.
+ for i in cfg'range loop
+ for j in cfg'range loop
+ if j > i then
+ -- both ranges are power of two aligned, so they overlap if
+ -- and only if the bases agree above the wider of the spans
+ wider := cfg(i).addr_span_bits;
+ if cfg(j).addr_span_bits > wider then
+ wider := cfg(j).addr_span_bits;
+ end if;
+ if ((cfg(i).base_addr xor cfg(j).base_addr) and not span_mask(wider)) = 32x"0" then
+ return false;
+ end if;
+ end if;
+ end loop;
+ end loop;
+ return true;
+ end function;
+
+ function bases_reachable (
+ constant cfg : axil_responder_cfg_array_t;
+ constant initiator_addr_width : integer
+ ) return boolean is
+ begin
+ -- Visit every responder and bail out on the first base the initiator
+ -- cannot drive, which is any base with a bit set at or above the
+ -- initiator's address width. Reusing span_mask here treats the initiator
+ -- width as a span: everything outside it must be zero.
+ for i in cfg'range loop
+ if (cfg(i).base_addr and not span_mask(initiator_addr_width)) /= 32x"0" then
+ return false;
+ end if;
+ end loop;
+ return true;
+ end function;
+
+end package body;
diff --git a/hdl/ip/vhd/axi_blocks/axil_interconnect_2k8.vhd b/hdl/ip/vhd/axi_blocks/axil_interconnect_2k8.vhd
index e155bd6a..6326972f 100644
--- a/hdl/ip/vhd/axi_blocks/axil_interconnect_2k8.vhd
+++ b/hdl/ip/vhd/axi_blocks/axil_interconnect_2k8.vhd
@@ -15,6 +15,11 @@ use work.axilite_if_2k8_pkg.all;
-- It is intended to be function as an MVP implementation allowing for basic multi-responder
-- usecases. It is not currently a full cross-bar implementation, but may grow to be one in the future.
-- This is the VHDL 2k8 version which does not use interface views.
+--
+-- Only one transaction, read or write, is in flight at a time: a registered
+-- decode stage selects a responder, the transaction runs to completion, and then
+-- the fabric is torn down and re-armed. Everything downstream of here relies on
+-- that, so it is worth stating plainly.
entity axil_interconnect_2k8 is
generic (
@@ -35,7 +40,7 @@ entity axil_interconnect_2k8 is
initiator_write_data_strb : in std_logic_vector(3 downto 0);
initiator_write_data_ready : out std_logic;
initiator_write_data_valid : in std_logic;
-
+
initiator_write_response_valid : out std_logic;
initiator_write_response_resp : out std_logic_vector(1 downto 0);
initiator_write_response_ready : in std_logic;
@@ -54,12 +59,12 @@ entity axil_interconnect_2k8 is
responders_write_address_valid : out std_logic_vector(config_array'range);
responders_write_address_ready : in std_logic_vector(config_array'range);
responders_write_address_addr : out tgt_addr32_t(config_array'range);
-
+
responders_write_data_valid : out std_logic_vector(config_array'range);
responders_write_data_ready : in std_logic_vector(config_array'range);
responders_write_data_data: out tgt_dat32_t(config_array'range);
responders_write_data_strb: out tgt_strb_t(config_array'range);
-
+
responders_write_response_ready : out std_logic_vector(config_array'range);
responders_write_response_resp : in tgt_resp_t(config_array'range);
responders_write_response_valid : in std_logic_vector(config_array'range);
@@ -78,134 +83,284 @@ end entity;
architecture rtl of axil_interconnect_2k8 is
- constant default_idx : integer := config_array'length;
- -- We implement a catch-all responder that will respond with an error if no other responder does
- -- so this signal is one larger than the number of responders
- signal responder_sel : integer range 0 to config_array'length := default_idx;
- signal responder_addr_width : integer range 0 to 32 := 8;
+ constant ZERO32 : std_logic_vector(31 downto 0) := (others => '0');
+
+ signal wr_addr32 : std_logic_vector(31 downto 0);
+ signal rd_addr32 : std_logic_vector(31 downto 0);
+ signal wr_hit : std_logic_vector(config_array'range);
+ signal rd_hit : std_logic_vector(config_array'range);
+
+ -- Registered select. One bit per responder, at most one set, plus a separate
+ -- bit for the catch-all error responder. Keeping this one-hot rather than an
+ -- integer index means the return path is a flat AND-OR tree instead of an
+ -- integer-to-one-hot decode buried in combinational logic.
+ signal sel_onehot : std_logic_vector(config_array'range);
+ signal sel_default : std_logic;
+ signal sel_is_write : std_logic;
+ signal in_txn : boolean;
+
+ -- A write request is only a request once both AW and W are on the bus, which
+ -- is the condition every responder applies before asserting AWREADY (see
+ -- axil_target_txn). Decoding on AWVALID alone is what let the FMC target's
+ -- spurious AWVALID re-assert issue a second write.
+ signal wr_req : std_logic;
+ signal rd_req : std_logic;
+
+ -- Set when a transaction completes with its request still asserted, and held
+ -- until the initiator drops it. Initiators here deassert VALID a cycle after
+ -- the handshake, so without this a stale request re-arms the fabric and a
+ -- duplicate transaction goes out behind the initiator's back.
+ signal wr_hold : std_logic;
+ signal rd_hold : std_logic;
+
signal write_done : std_logic;
signal read_done : std_logic;
- signal in_txn : boolean;
+
+ -- Fabric side of each responder's optional pipeline. With pipe_stages = 0
+ -- these are wired straight through to the responder ports.
+ signal mux_write_address_valid : std_logic_vector(config_array'range);
+ signal mux_write_address_ready : std_logic_vector(config_array'range);
+ signal mux_write_address_addr : tgt_addr32_t(config_array'range);
+ signal mux_write_data_valid : std_logic_vector(config_array'range);
+ signal mux_write_data_ready : std_logic_vector(config_array'range);
+ signal mux_write_data_data : tgt_dat32_t(config_array'range);
+ signal mux_write_data_strb : tgt_strb_t(config_array'range);
+ signal mux_write_response_valid : std_logic_vector(config_array'range);
+ signal mux_write_response_ready : std_logic_vector(config_array'range);
+ signal mux_write_response_resp : tgt_resp_t(config_array'range);
+ signal mux_read_address_valid : std_logic_vector(config_array'range);
+ signal mux_read_address_ready : std_logic_vector(config_array'range);
+ signal mux_read_address_addr : tgt_addr32_t(config_array'range);
+ signal mux_read_data_valid : std_logic_vector(config_array'range);
+ signal mux_read_data_ready : std_logic_vector(config_array'range);
+ signal mux_read_data_data : tgt_dat32_t(config_array'range);
+ signal mux_read_data_resp : tgt_resp_t(config_array'range);
begin
+ assert config_array'low = 0
+ report "config_array must be indexed from 0"
+ severity failure;
+ assert bases_aligned(config_array)
+ report "every responder base address must be aligned to its own addr_span_bits"
+ severity failure;
+ assert ranges_disjoint(config_array)
+ report "responder address ranges must not overlap"
+ severity failure;
+ assert bases_reachable(config_array, initiator_addr_width)
+ report "a responder base address is outside the initiator's address space"
+ severity failure;
+
+ wr_addr32 <= resize(initiator_write_address_addr, 32);
+ rd_addr32 <= resize(initiator_read_address_addr, 32);
+
+ wr_req <= initiator_write_address_valid and initiator_write_data_valid;
+ rd_req <= initiator_read_address_valid;
+
write_done <= '1' when initiator_write_response_valid = '1' and initiator_write_response_ready = '1' else
'0';
read_done <= '1' when initiator_read_data_valid = '1' and initiator_read_data_ready = '1' else
'0';
- -- we're going to stall all the transactions until we have decoded and selected a responder,
- -- flipped the muxes and then we can let the txn_through, and we keep the responder selected until
- -- the txn is done.
- -- There's a lot of combo logic here, we'll see how this goes.
+ -- Address decode. Spans are powers of two and bases are aligned to their own
+ -- span (asserted above), so "inside the range" is just "the bits above the
+ -- span match the base". The masked-off low bits fold away in synthesis,
+ -- leaving an equality compare on the upper bits rather than the pair of
+ -- magnitude compares, and their carry chains, this used to build. Comparing
+ -- over the full 32 bits matters: a narrow initiator must not match a base
+ -- that it cannot actually address, and the extra bits are constant zeros.
+ hit_gen: for i in config_array'range generate
+ constant span : natural := config_array(i).addr_span_bits;
+ begin
+
+ wr_hit(i) <= '1' when ((wr_addr32 xor config_array(i).base_addr) and not span_mask(span)) = ZERO32 else
+ '0';
+ rd_hit(i) <= '1' when ((rd_addr32 xor config_array(i).base_addr) and not span_mask(span)) = ZERO32 else
+ '0';
+
+ end generate;
+
+ -- Stall the transaction for one cycle while the responder is selected, then
+ -- hold that selection until the transaction completes.
decode: process(clk, reset)
begin
if reset = '1' then
- responder_sel <= default_idx;
- responder_addr_width <= 8;
+ sel_onehot <= (others => '0');
+ sel_default <= '0';
+ sel_is_write <= '0';
in_txn <= false;
+ wr_hold <= '0';
+ rd_hold <= '0';
elsif rising_edge(clk) then
- if initiator_write_address_valid = '1' then
- for i in 0 to config_array'length - 1 loop
- if (initiator_write_address_addr >= config_array(i).base_addr) and
- (initiator_write_address_addr < config_array(i).base_addr + 2**config_array(i).addr_span_bits) then
- responder_sel <= i;
- responder_addr_width <= config_array(i).addr_span_bits;
- end if;
- end loop;
- in_txn <= true;
- elsif initiator_read_address_valid = '1' then
- for i in 0 to config_array'length - 1 loop
- if (initiator_read_address_addr >= config_array(i).base_addr) and
- (initiator_read_address_addr < config_array(i).base_addr + 2**config_array(i).addr_span_bits) then
- responder_sel <= i;
- responder_addr_width <= config_array(i).addr_span_bits;
- end if;
- end loop;
- in_txn <= true;
- elsif write_done or read_done then
- responder_sel <= default_idx;
+ -- Per channel re-arm guard, tracked independently so a permanently
+ -- asserted AWVALID cannot block reads.
+ if write_done = '1' then
+ wr_hold <= wr_req;
+ elsif wr_hold = '1' and wr_req = '0' then
+ wr_hold <= '0';
+ end if;
+
+ if read_done = '1' then
+ rd_hold <= rd_req;
+ elsif rd_hold = '1' and rd_req = '0' then
+ rd_hold <= '0';
+ end if;
+
+ -- Teardown takes priority over arming. An initiator that leaves
+ -- AWVALID asserted past the end of its write must not be able to
+ -- pin the fabric to the previous selection.
+ if write_done = '1' or read_done = '1' then
+ sel_onehot <= (others => '0');
+ sel_default <= '0';
+ sel_is_write <= '0';
in_txn <= false;
- responder_addr_width <= 8;
+ elsif not in_txn then
+ if wr_req = '1' and wr_hold = '0' then
+ sel_onehot <= wr_hit;
+ sel_default <= not (or wr_hit);
+ sel_is_write <= '1';
+ in_txn <= true;
+ elsif rd_req = '1' and rd_hold = '0' then
+ sel_onehot <= rd_hit;
+ sel_default <= not (or rd_hit);
+ sel_is_write <= '0';
+ in_txn <= true;
+ end if;
end if;
end if;
end process;
- mux: process(all)
- variable masked_addr : std_logic_vector(31 downto 0);
+ -- Return path: a one-hot AND-OR mux of the selected responder, or the
+ -- catch-all error responder when nothing matched.
+ ret_mux: process(all)
+ variable awready : std_logic;
+ variable wready : std_logic;
+ variable bvalid : std_logic;
+ variable bresp : std_logic_vector(1 downto 0);
+ variable arready : std_logic;
+ variable rvalid : std_logic;
+ variable rresp : std_logic_vector(1 downto 0);
+ variable rdata : std_logic_vector(31 downto 0);
begin
- -- default no transaction state for all responders
- responders_write_address_addr <= (others => (others => '0'));
- responders_read_address_addr <= (others => (others => '0'));
- masked_addr := resize(initiator_write_address_addr, 32);
- for i in 31 downto 0 loop
- if i >= responder_addr_width or i >= initiator_write_address_addr'length then
- masked_addr(i) := '0';
- else
- masked_addr(i) := initiator_write_address_addr(i);
- end if;
- end loop;
- for i in 0 to config_array'length - 1 loop
- responders_write_address_valid(i) <= '0';
- responders_write_data_valid(i) <= '0';
- responders_write_data_data(i) <= initiator_write_data_data;
- responders_write_data_strb(i) <= initiator_write_data_strb;
- responders_write_address_addr(i) <= masked_addr;
- responders_write_response_ready(i) <= '0';
-
- responders_read_address_valid(i) <= '0';
-
- responders_read_address_addr(i)<= masked_addr;
- responders_read_data_ready(i) <= '0';
+ awready := '0';
+ wready := '0';
+ bvalid := '0';
+ bresp := SLVERR;
+ arready := '0';
+ rvalid := '0';
+ rresp := SLVERR;
+ rdata := (others => '0');
+ for i in config_array'range loop
+ if sel_onehot(i) = '1' then
+ awready := mux_write_address_ready(i);
+ wready := mux_write_data_ready(i);
+ bvalid := mux_write_response_valid(i);
+ bresp := mux_write_response_resp(i);
+ arready := mux_read_address_ready(i);
+ rvalid := mux_read_data_valid(i);
+ rresp := mux_read_data_resp(i);
+ rdata := mux_read_data_data(i);
+ end if;
end loop;
- -- deal with in-txn muxing
- if in_txn and responder_sel < default_idx then
-
- -- responder mux
- -- we already assigned addresses to the responder addresss above, no need to overwrite here
- responders_write_address_valid(responder_sel) <= initiator_write_address_valid;
- responders_write_data_valid(responder_sel) <= initiator_write_data_valid;
- responders_write_data_data(responder_sel) <= initiator_write_data_data;
- responders_write_data_strb(responder_sel) <= initiator_write_data_strb;
- responders_write_response_ready(responder_sel) <= initiator_write_response_ready;
-
- responders_read_address_valid(responder_sel) <= initiator_read_address_valid;
- responders_read_data_ready(responder_sel) <= initiator_read_data_ready;
- -- initiator mux
- initiator_write_address_ready <= responders_write_address_ready(responder_sel);
- initiator_write_data_ready <= responders_write_data_ready(responder_sel);
- initiator_write_response_resp <= responders_write_response_resp(responder_sel);
- initiator_write_response_valid <= responders_write_response_valid(responder_sel);
- initiator_read_address_ready <= responders_read_address_ready(responder_sel);
- initiator_read_data_resp <= responders_read_data_resp(responder_sel);
- initiator_read_data_valid <= responders_read_data_valid(responder_sel);
- initiator_read_data_data <= responders_read_data_data(responder_sel);
-
- elsif in_txn then
- -- default response to not hang bus
- initiator_write_address_ready <= '1';
- initiator_write_data_ready <= '1';
- initiator_write_response_resp <= SLVERR;
- initiator_write_response_valid <= '1';
- initiator_read_address_ready <= '1';
- initiator_read_data_resp <= SLVERR;
- initiator_read_data_valid <= '1';
- initiator_read_data_data <= X"DEADBEEF";
-
- else
- -- hold for decode
- -- default response to not hang bus
- initiator_write_address_ready <= '0';
- initiator_write_data_ready <= '0';
- initiator_write_response_resp <= SLVERR;
- initiator_write_response_valid <= '0';
- initiator_read_address_ready <= '0';
- initiator_read_data_resp <= SLVERR;
- initiator_read_data_valid <= '0';
- initiator_read_data_data <= (others => '0');
+ if sel_default = '1' then
+ -- Nothing decoded, so answer immediately with an error rather than
+ -- hanging the bus. Only the channel that was actually decoded is
+ -- answered, so an unmapped read cannot hand an AWREADY to a write
+ -- that has not presented its data yet.
+ if sel_is_write = '1' then
+ awready := '1';
+ wready := '1';
+ bvalid := '1';
+ bresp := SLVERR;
+ else
+ arready := '1';
+ rvalid := '1';
+ rresp := SLVERR;
+ rdata := X"DEADBEEF";
+ end if;
end if;
+
+ initiator_write_address_ready <= awready;
+ initiator_write_data_ready <= wready;
+ initiator_write_response_valid <= bvalid;
+ initiator_write_response_resp <= bresp;
+ initiator_read_address_ready <= arready;
+ initiator_read_data_valid <= rvalid;
+ initiator_read_data_resp <= rresp;
+ initiator_read_data_data <= rdata;
end process;
-
-end rtl;
\ No newline at end of file
+
+ -- Forward path. The payload is broadcast to every responder, masked to that
+ -- responder's own span, and only the selected responder sees a valid. The
+ -- masked-off address bits are hard constant zeros, so the fabric side of
+ -- each responder's address bus collapses to span real wires.
+ --
+ -- Each responder then optionally goes through axil_pipe, which inserts
+ -- config_array(i).pipe_stages register stages in each direction so a
+ -- responder that sits a long way from the fabric does not have to be reached
+ -- and answered within one clock period. pipe_stages = 0 is a pass-through and
+ -- costs nothing.
+ resp_gen: for i in config_array'range generate
+ constant span : natural := config_array(i).addr_span_bits;
+ begin
+
+ mux_write_address_addr(i) <= wr_addr32 and span_mask(span);
+ mux_read_address_addr(i) <= rd_addr32 and span_mask(span);
+ mux_write_data_data(i) <= initiator_write_data_data;
+ mux_write_data_strb(i) <= initiator_write_data_strb;
+
+ mux_write_address_valid(i) <= initiator_write_address_valid and sel_onehot(i);
+ mux_write_data_valid(i) <= initiator_write_data_valid and sel_onehot(i);
+ mux_write_response_ready(i) <= initiator_write_response_ready and sel_onehot(i);
+ mux_read_address_valid(i) <= initiator_read_address_valid and sel_onehot(i);
+ mux_read_data_ready(i) <= initiator_read_data_ready and sel_onehot(i);
+
+ axil_pipe_inst: entity work.axil_pipe
+ generic map (
+ stages => config_array(i).pipe_stages,
+ addr_width => span
+ )
+ port map (
+ clk => clk,
+ reset => reset,
+ sink_awaddr => mux_write_address_addr(i),
+ sink_awvalid => mux_write_address_valid(i),
+ sink_awready => mux_write_address_ready(i),
+ sink_wdata => mux_write_data_data(i),
+ sink_wstrb => mux_write_data_strb(i),
+ sink_wvalid => mux_write_data_valid(i),
+ sink_wready => mux_write_data_ready(i),
+ sink_bvalid => mux_write_response_valid(i),
+ sink_bresp => mux_write_response_resp(i),
+ sink_bready => mux_write_response_ready(i),
+ sink_araddr => mux_read_address_addr(i),
+ sink_arvalid => mux_read_address_valid(i),
+ sink_arready => mux_read_address_ready(i),
+ sink_rvalid => mux_read_data_valid(i),
+ sink_rdata => mux_read_data_data(i),
+ sink_rresp => mux_read_data_resp(i),
+ sink_rready => mux_read_data_ready(i),
+ source_awaddr => responders_write_address_addr(i),
+ source_awvalid => responders_write_address_valid(i),
+ source_awready => responders_write_address_ready(i),
+ source_wdata => responders_write_data_data(i),
+ source_wstrb => responders_write_data_strb(i),
+ source_wvalid => responders_write_data_valid(i),
+ source_wready => responders_write_data_ready(i),
+ source_bvalid => responders_write_response_valid(i),
+ source_bresp => responders_write_response_resp(i),
+ source_bready => responders_write_response_ready(i),
+ source_araddr => responders_read_address_addr(i),
+ source_arvalid => responders_read_address_valid(i),
+ source_arready => responders_read_address_ready(i),
+ source_rvalid => responders_read_data_valid(i),
+ source_rdata => responders_read_data_data(i),
+ source_rresp => responders_read_data_resp(i),
+ source_rready => responders_read_data_ready(i)
+ );
+
+ end generate;
+
+end rtl;
diff --git a/hdl/ip/vhd/axi_blocks/axil_pipe.vhd b/hdl/ip/vhd/axi_blocks/axil_pipe.vhd
new file mode 100644
index 00000000..d633dac0
--- /dev/null
+++ b/hdl/ip/vhd/axi_blocks/axil_pipe.vhd
@@ -0,0 +1,383 @@
+-- This Source Code Form is subject to the terms of the Mozilla Public
+-- License, v. 2.0. If a copy of the MPL was not distributed with this
+-- file, You can obtain one at https://mozilla.org/MPL/2.0/.
+
+
+library ieee;
+use ieee.std_logic_1164.all;
+use ieee.numeric_std.all;
+
+use work.axil_common_pkg.all;
+
+-- Configurable pipeline stages between the interconnect fabric and one
+-- responder, so a responder that is physically far from the fabric does not have
+-- to be reached and answered inside a single clock period.
+--
+-- The fabric admits one transaction, read or write, at a time. That means this
+-- does not have to be five independent AXI channel register slices: the whole
+-- transaction serializes into one request bundle going out and one response
+-- bundle coming back, which is roughly a third of the flops a pair of full
+-- register slices would cost, with a single token chain instead of five sets of
+-- valid/ready control.
+--
+-- Each payload stage advances only behind its own token, so every stage holds
+-- what it was given until the next transaction pushes through it. That is what
+-- keeps the far end of the chain stable across a multi-cycle handshake, and it is
+-- why no separate capture register is needed at either end. Gating the whole
+-- chain on the OR of all the tokens instead would clobber the last stage on the
+-- cycle the token reached it.
+--
+-- The single-outstanding-transaction property of the fabric is load bearing here:
+-- it is what makes it safe for a stage to hold its payload indefinitely, and what
+-- guarantees a new request cannot enter the chain while a response is still on
+-- its way back.
+--
+-- Round trip latency is 2 * stages cycles plus a handful of fixed handshake
+-- cycles. stages = 0 is a plain pass-through and costs nothing.
+
+entity axil_pipe is
+ generic (
+ -- Register stages inserted in *each* direction
+ stages : natural;
+ -- Address bits this responder actually decodes
+ addr_width : natural
+ );
+ port (
+ clk : in std_logic;
+ reset : in std_logic;
+
+ -- Fabric facing side, a *target* interface
+ sink_awaddr : in std_logic_vector(31 downto 0);
+ sink_awvalid : in std_logic;
+ sink_awready : out std_logic;
+
+ sink_wdata : in std_logic_vector(31 downto 0);
+ sink_wstrb : in std_logic_vector(3 downto 0);
+ sink_wvalid : in std_logic;
+ sink_wready : out std_logic;
+
+ sink_bvalid : out std_logic;
+ sink_bresp : out std_logic_vector(1 downto 0);
+ sink_bready : in std_logic;
+
+ sink_araddr : in std_logic_vector(31 downto 0);
+ sink_arvalid : in std_logic;
+ sink_arready : out std_logic;
+
+ sink_rvalid : out std_logic;
+ sink_rdata : out std_logic_vector(31 downto 0);
+ sink_rresp : out std_logic_vector(1 downto 0);
+ sink_rready : in std_logic;
+
+ -- Responder facing side, a *controller* interface
+ source_awaddr : out std_logic_vector(31 downto 0);
+ source_awvalid : out std_logic;
+ source_awready : in std_logic;
+
+ source_wdata : out std_logic_vector(31 downto 0);
+ source_wstrb : out std_logic_vector(3 downto 0);
+ source_wvalid : out std_logic;
+ source_wready : in std_logic;
+
+ source_bvalid : in std_logic;
+ source_bresp : in std_logic_vector(1 downto 0);
+ source_bready : out std_logic;
+
+ source_araddr : out std_logic_vector(31 downto 0);
+ source_arvalid : out std_logic;
+ source_arready : in std_logic;
+
+ source_rvalid : in std_logic;
+ source_rdata : in std_logic_vector(31 downto 0);
+ source_rresp : in std_logic_vector(1 downto 0);
+ source_rready : out std_logic
+ );
+end entity;
+
+architecture rtl of axil_pipe is
+
+begin
+
+ pipe_gen: if stages > 0 generate
+
+ -- request bundle: is_write & addr & wdata & wstrb
+ constant STRB_LO : natural := 0;
+ constant STRB_HI : natural := 3;
+ constant DATA_LO : natural := 4;
+ constant DATA_HI : natural := 35;
+ constant ADDR_LO : natural := 36;
+ constant ADDR_HI : natural := 36 + addr_width - 1;
+ constant IS_WRITE : natural := ADDR_HI + 1;
+ constant REQ_W : natural := IS_WRITE + 1;
+
+ -- response bundle: rdata & resp
+ constant RESP_LO : natural := 0;
+ constant RESP_HI : natural := 1;
+ constant RDATA_LO : natural := 2;
+ constant RDATA_HI : natural := 33;
+ constant RSP_W : natural := RDATA_HI + 1;
+
+ constant LAST : natural := stages - 1;
+
+ type req_sr_t is array (0 to stages - 1) of std_logic_vector(REQ_W - 1 downto 0);
+ type rsp_sr_t is array (0 to stages - 1) of std_logic_vector(RSP_W - 1 downto 0);
+
+ -- deliberately not reset: only the tokens need to come up clean, and
+ -- leaving the payload chain resetless lets Vivado map it to SRLs
+ signal req_sr : req_sr_t;
+ signal rsp_sr : rsp_sr_t;
+
+ signal fwd_tok : std_logic_vector(stages - 1 downto 0);
+ signal ret_tok : std_logic_vector(stages - 1 downto 0);
+
+ type sink_state_t is (armed, wait_rsp, respond, request_gap);
+
+ signal sink_state : sink_state_t;
+ signal sink_is_write : std_logic;
+ -- registered one-shot, so AW and W always handshake in the same cycle
+ -- and AWREADY never depends combinationally on AWVALID
+ signal sink_wr_ack : std_logic;
+ signal sink_rd_ack : std_logic;
+ signal sink_resp_valid : std_logic;
+
+ type source_state_t is (idle, issue);
+
+ signal source_state : source_state_t;
+ signal src_is_write : std_logic;
+ signal src_awvalid : std_logic;
+ signal src_wvalid : std_logic;
+ signal src_arvalid : std_logic;
+ signal src_bready : std_logic;
+ signal src_rready : std_logic;
+ signal src_aw_done : std_logic;
+ signal src_w_done : std_logic;
+ signal src_b_done : std_logic;
+ signal src_ar_done : std_logic;
+ signal src_r_done : std_logic;
+
+ begin
+
+ sink_awready <= sink_wr_ack;
+ sink_wready <= sink_wr_ack;
+ sink_arready <= sink_rd_ack;
+ sink_bvalid <= sink_resp_valid and sink_is_write;
+ sink_rvalid <= sink_resp_valid and not sink_is_write;
+ sink_bresp <= rsp_sr(LAST)(RESP_HI downto RESP_LO);
+ sink_rresp <= rsp_sr(LAST)(RESP_HI downto RESP_LO);
+ sink_rdata <= rsp_sr(LAST)(RDATA_HI downto RDATA_LO);
+
+ source_awvalid <= src_awvalid;
+ source_wvalid <= src_wvalid;
+ source_arvalid <= src_arvalid;
+ source_bready <= src_bready;
+ source_rready <= src_rready;
+ source_wdata <= req_sr(LAST)(DATA_HI downto DATA_LO);
+ source_wstrb <= req_sr(LAST)(STRB_HI downto STRB_LO);
+ -- the fabric already masked the address to this responder's span, so the
+ -- bits above it are constant zeros
+ source_awaddr <= (31 downto addr_width => '0') & req_sr(LAST)(ADDR_HI downto ADDR_LO);
+ source_araddr <= (31 downto addr_width => '0') & req_sr(LAST)(ADDR_HI downto ADDR_LO);
+
+ -- Outbound: accept a transaction from the fabric, walk it down to the
+ -- responder, and present the response that comes back.
+ fwd: process(clk, reset)
+ begin
+ if reset = '1' then
+ sink_state <= armed;
+ sink_is_write <= '0';
+ sink_wr_ack <= '0';
+ sink_rd_ack <= '0';
+ sink_resp_valid <= '0';
+ fwd_tok <= (others => '0');
+ elsif rising_edge(clk) then
+ sink_wr_ack <= '0';
+ sink_rd_ack <= '0';
+
+ for j in stages - 1 downto 1 loop
+ fwd_tok(j) <= fwd_tok(j - 1);
+ if fwd_tok(j - 1) = '1' then
+ req_sr(j) <= req_sr(j - 1);
+ end if;
+ end loop;
+ fwd_tok(0) <= '0';
+
+ case sink_state is
+ when armed =>
+ -- A write is only taken once AW and W are both present,
+ -- which is what every responder in the tree requires
+ -- before asserting AWREADY. WDATA has to be captured
+ -- here: the FMC target's write data FIFO pops on the W
+ -- handshake, so it is not valid afterwards.
+ if sink_awvalid = '1' and sink_wvalid = '1' then
+ req_sr(0)(IS_WRITE) <= '1';
+ req_sr(0)(ADDR_HI downto ADDR_LO) <= sink_awaddr(addr_width - 1 downto 0);
+ req_sr(0)(DATA_HI downto DATA_LO) <= sink_wdata;
+ req_sr(0)(STRB_HI downto STRB_LO) <= sink_wstrb;
+ fwd_tok(0) <= '1';
+ sink_wr_ack <= '1';
+ sink_is_write <= '1';
+ sink_state <= wait_rsp;
+ elsif sink_arvalid = '1' then
+ req_sr(0)(IS_WRITE) <= '0';
+ req_sr(0)(ADDR_HI downto ADDR_LO) <= sink_araddr(addr_width - 1 downto 0);
+ fwd_tok(0) <= '1';
+ sink_rd_ack <= '1';
+ sink_is_write <= '0';
+ sink_state <= wait_rsp;
+ end if;
+
+ when wait_rsp =>
+ if ret_tok(LAST) = '1' then
+ sink_resp_valid <= '1';
+ sink_state <= respond;
+ end if;
+
+ when respond =>
+ if (sink_is_write = '1' and sink_bready = '1') or
+ (sink_is_write = '0' and sink_rready = '1') then
+ sink_resp_valid <= '0';
+ sink_state <= request_gap;
+ end if;
+
+ when request_gap =>
+ -- Do not re-arm until the request that just completed is
+ -- off the bus. Initiators here deassert VALID a cycle
+ -- after the handshake, so without this a stale request
+ -- would be issued a second time.
+ if (sink_is_write = '1' and not (sink_awvalid = '1' and sink_wvalid = '1')) or
+ (sink_is_write = '0' and sink_arvalid = '0') then
+ sink_state <= armed;
+ end if;
+
+ end case;
+ end if;
+ end process;
+
+ -- Inbound: run the transaction against the responder and walk the
+ -- response back up. AW and W are tracked separately because responders
+ -- are allowed to accept them independently, and the handshakes are all
+ -- sampled in one state because axil_target_txn presents BVALID as a
+ -- single cycle pulse when BREADY is already asserted and ARREADY
+ -- combinationally, so a dedicated wait-for-response state would miss
+ -- them.
+ ret: process(clk, reset)
+ variable aw_done : std_logic;
+ variable w_done : std_logic;
+ variable b_done : std_logic;
+ variable ar_done : std_logic;
+ variable r_done : std_logic;
+ begin
+ if reset = '1' then
+ source_state <= idle;
+ src_is_write <= '0';
+ src_awvalid <= '0';
+ src_wvalid <= '0';
+ src_arvalid <= '0';
+ src_bready <= '0';
+ src_rready <= '0';
+ src_aw_done <= '0';
+ src_w_done <= '0';
+ src_b_done <= '0';
+ src_ar_done <= '0';
+ src_r_done <= '0';
+ ret_tok <= (others => '0');
+ elsif rising_edge(clk) then
+ for j in stages - 1 downto 1 loop
+ ret_tok(j) <= ret_tok(j - 1);
+ if ret_tok(j - 1) = '1' then
+ rsp_sr(j) <= rsp_sr(j - 1);
+ end if;
+ end loop;
+ ret_tok(0) <= '0';
+
+ case source_state is
+ when idle =>
+ if fwd_tok(LAST) = '1' then
+ src_is_write <= req_sr(LAST)(IS_WRITE);
+ if req_sr(LAST)(IS_WRITE) = '1' then
+ src_awvalid <= '1';
+ src_wvalid <= '1';
+ src_bready <= '1';
+ else
+ src_arvalid <= '1';
+ src_rready <= '1';
+ end if;
+ src_aw_done <= '0';
+ src_w_done <= '0';
+ src_b_done <= '0';
+ src_ar_done <= '0';
+ src_r_done <= '0';
+ source_state <= issue;
+ end if;
+
+ when issue =>
+ aw_done := src_aw_done;
+ w_done := src_w_done;
+ b_done := src_b_done;
+ ar_done := src_ar_done;
+ r_done := src_r_done;
+
+ if src_awvalid = '1' and source_awready = '1' then
+ src_awvalid <= '0';
+ aw_done := '1';
+ end if;
+ if src_wvalid = '1' and source_wready = '1' then
+ src_wvalid <= '0';
+ w_done := '1';
+ end if;
+ if source_bvalid = '1' and src_bready = '1' then
+ b_done := '1';
+ rsp_sr(0)(RESP_HI downto RESP_LO) <= source_bresp;
+ end if;
+ if src_arvalid = '1' and source_arready = '1' then
+ src_arvalid <= '0';
+ ar_done := '1';
+ end if;
+ if source_rvalid = '1' and src_rready = '1' then
+ r_done := '1';
+ rsp_sr(0)(RDATA_HI downto RDATA_LO) <= source_rdata;
+ rsp_sr(0)(RESP_HI downto RESP_LO) <= source_rresp;
+ end if;
+
+ src_aw_done <= aw_done;
+ src_w_done <= w_done;
+ src_b_done <= b_done;
+ src_ar_done <= ar_done;
+ src_r_done <= r_done;
+
+ if (src_is_write = '1' and aw_done = '1' and w_done = '1' and b_done = '1') or
+ (src_is_write = '0' and ar_done = '1' and r_done = '1') then
+ ret_tok(0) <= '1';
+ src_bready <= '0';
+ src_rready <= '0';
+ source_state <= idle;
+ end if;
+
+ end case;
+ end if;
+ end process;
+
+ else generate
+
+ sink_awready <= source_awready;
+ sink_wready <= source_wready;
+ sink_bvalid <= source_bvalid;
+ sink_bresp <= source_bresp;
+ sink_arready <= source_arready;
+ sink_rvalid <= source_rvalid;
+ sink_rdata <= source_rdata;
+ sink_rresp <= source_rresp;
+
+ source_awaddr <= sink_awaddr;
+ source_awvalid <= sink_awvalid;
+ source_wdata <= sink_wdata;
+ source_wstrb <= sink_wstrb;
+ source_wvalid <= sink_wvalid;
+ source_bready <= sink_bready;
+ source_araddr <= sink_araddr;
+ source_arvalid <= sink_arvalid;
+ source_rready <= sink_rready;
+
+ end generate;
+
+end rtl;
diff --git a/hdl/ip/vhd/axi_blocks/docs/axil_interconnect.adoc b/hdl/ip/vhd/axi_blocks/docs/axil_interconnect.adoc
new file mode 100644
index 00000000..c3522068
--- /dev/null
+++ b/hdl/ip/vhd/axi_blocks/docs/axil_interconnect.adoc
@@ -0,0 +1,219 @@
+:showtitle:
+:toc: left
+:numbered:
+:icons: font
+:revision: 1.0
+:revdate: 2026-08-05
+
+= AXI-Lite Interconnect
+
+`axil_interconnect_2k8` is the shared register-access fabric used by cosmo_seq, grapefruit and
+cosmo_hp. One initiator (the FMC target on the Spartan-7 designs, the SPI AXI controller on
+cosmo_hp) reaches a handful of responders, each of which owns a power-of-two slice of the address
+space.
+
+`axil_interconnect` is a thin VHDL-2019 wrapper that presents the same thing with interface views;
+all of the logic lives in the 2008 flat-port entity so that cosmo_hp, which builds through GHDL and
+yosys, can use it too.
+
+== Overview
+image::axil_interconnect_block.drawio.svg[align="center"]
+
+The single most important property of this block, and the one everything else leans on:
+
+[IMPORTANT]
+====
+Exactly one transaction, read *or* write, is in flight at a time. A registered decode stage selects
+a responder, the transaction runs to completion, and only then is the fabric torn down and re-armed.
+====
+
+Both initiators cooperate with this — each dispatches a read or a write from an idle state and does
+not return to idle until B or R has landed, and neither ever asserts `AWVALID` and `ARVALID`
+together.
+
+=== Address decode
+
+Every responder base address is aligned to its own span, and every span is a power of two. That
+makes "is this address inside the range" equivalent to "do the address bits above the span match the
+base", which is a plain equality compare rather than the pair of magnitude compares (and their carry
+chains) that a base/limit check would build.
+
+[source,vhdl]
+----
+wr_hit(i) <= '1' when ((wr_addr32 xor config_array(i).base_addr) and not span_mask(span)) = ZERO32
+ else '0';
+----
+
+The compare is deliberately over the full 32 bits of a *resized* address. Narrowing it to the
+initiator's own width would let a 16-bit initiator falsely match a base it has no way to drive; the
+extra bits compare against constant zeros and fold away in synthesis.
+
+The invariants this relies on are checked at elaboration by `bases_aligned`, `ranges_disjoint` and
+`bases_reachable` in `axil_common_pkg`, so a bad map is a build error rather than a silent misroute.
+
+=== Responder select
+
+The select is a registered *one-hot* vector, `sel_onehot`, one bit per responder, plus a separate
+`sel_default` bit for the catch-all. Keeping it one-hot rather than an integer index means the
+return path is a flat AND-OR tree; an integer index forces synthesis to build an
+integer-to-one-hot decoder inside the combinational mux.
+
+A write is only decoded once `AWVALID` *and* `WVALID` are both on the bus. That is the same
+condition every responder already applies before asserting `AWREADY` (see `axil_target_txn`), and it
+keeps a lone `AWVALID` from arming the fabric.
+
+Teardown takes priority over arming, and re-arming is blocked per channel until the initiator drops
+the request it just completed. Initiators here deassert VALID one cycle *after* the handshake, so
+without that guard a stale request re-arms the fabric and a duplicate transaction goes out behind
+the initiator's back. The guard is tracked separately for reads and writes so that a permanently
+asserted `AWVALID` cannot block reads.
+
+=== Error responder
+
+When no responder matches, `sel_default` is set instead of any bit of `sel_onehot`, and the fabric
+answers by itself: `SLVERR`, with `RDATA` of `0xDEADBEEF` on reads. It answers only the channel that
+was actually decoded, tracked by `sel_is_write`, so an unmapped read cannot hand an `AWREADY` to a
+write that has not presented its data yet.
+
+Because the response is registered, an unmapped access costs the same one cycle of decode plus one
+cycle of response as a mapped access to a fast responder. It never stalls, which is the entire point
+of having it.
+
+== Pipelining
+
+Adding pipeline stages is how a responder that sits a long way from the fabric stops having to be
+reached *and* answered inside a single clock period. Each responder gets its own depth, set by
+`pipe_stages` in the config record:
+
+[source,vhdl]
+----
+constant config_array : axil_responder_cfg_array_t :=
+ (INFO_RESP_IDX => resp_cfg(base_addr => x"00000000", addr_span_bits => 8),
+ ...
+ ESPI_RESP_IDX => resp_cfg(base_addr => x"00008000", addr_span_bits => 15, pipe_stages => 1));
+----
+
+`pipe_stages = 0` generates a plain pass-through and costs nothing, so the knob is free where it is
+left alone.
+
+image::axil_pipe.drawio.svg[align="center"]
+
+=== Why it is not five register slices
+
+The obvious implementation is a register slice per AXI channel, which for AW/W/B/AR/R means five
+independent sets of valid/ready control and roughly 292 flops per stage per responder.
+
+Because the fabric admits one transaction at a time, `axil_pipe` does not need that. The whole
+transaction serializes into *one* request bundle going out and *one* response bundle coming back:
+
+[cols="1,3"]
+|===
+| request bundle | `is_write & addr(addr_width) & wdata(32) & wstrb(4)`
+| response bundle | `rdata(32) & resp(2)`
+|===
+
+Only the address bits the responder actually decodes are carried, so an 8-bit responder pipes 45
+bits rather than a 32-bit address. That lands at roughly a third of the flops of a pair of full
+register slices, with a single 1-bit token chain instead of five sets of handshake control.
+
+Deliberately, no timing exceptions are required. cosmo_hp builds through yosys and nextpnr, where
+there is no way to express a `set_multicycle_path`, so an approach that leaned on constraints would
+have bought nothing on the one design whose critical path was actually in this block.
+
+=== How the chain advances
+
+Each payload stage advances *only behind its own token*:
+
+[source,vhdl]
+----
+for j in stages - 1 downto 1 loop
+ fwd_tok(j) <= fwd_tok(j - 1);
+ if fwd_tok(j - 1) = '1' then
+ req_sr(j) <= req_sr(j - 1);
+ end if;
+end loop;
+----
+
+That is what makes every stage hold its payload until the next transaction pushes through it, which
+in turn keeps the far end of the chain stable across a multi-cycle handshake — and that is why no
+separate capture register is needed at either end. Stage 0 *is* the sink's capture register and
+stage `stages-1` *is* the source's.
+
+[WARNING]
+====
+Gating the whole chain on the OR of all the tokens instead would clobber the last stage on the cycle
+the token reached it, corrupting the payload the source FSM is still driving.
+====
+
+The payload chain is deliberately left resetless — only the tokens come up clean — which on 7-series
+also lets Vivado map it to SRLs.
+
+=== The two ends
+
+The *sink* FSM (`armed`, `wait_rsp`, `respond`, `request_gap`) faces the fabric. It accepts a write
+only when `AWVALID` and `WVALID` are both present and asserts `AWREADY` and `WREADY` together in the
+same cycle, as a registered one-shot — never combinationally off `AWVALID`, which would create a
+path straight back to the initiator that does not exist today. It captures `WDATA` on acceptance
+because the FMC target's write-data FIFO pops on the W handshake and the data is not valid
+afterwards. It will not re-arm until the request that just completed is off the bus.
+
+The *source* FSM (`idle`, `issue`) faces the responder. It tracks the AW and W handshakes separately,
+because responders are free to accept them independently, and samples every handshake in one state.
+That last part matters: `axil_target_txn` presents `BVALID` as a single-cycle pulse when `BREADY` is
+already asserted, and `ARREADY` combinationally as `not rvalid`, so a dedicated wait-for-response
+state would miss both and hang.
+
+=== Latency
+
+Round trip is `2 * pipe_stages` cycles plus a few fixed handshake cycles. Measured by the
+`pipe_latency` testbench case, as clocks from `ARVALID` to `RVALID`:
+
+[cols="1,1"]
+|===
+| `pipe_stages` | read latency
+
+| 0 | 2 clocks
+| 1 | 6 clocks
+| 2 | 8 clocks
+|===
+
+Each additional stage costs exactly 2 clocks; the step from 0 to 1 also picks up the pipe's fixed
+handshake overhead. At 125 MHz one stage is about 40 ns of extra SP bus stall per access, which the
+FMC covers with wait states — but it is a real cost, so raise `pipe_stages` only where timing needs
+it and re-measure.
+
+== Simulation
+
+`buck2 run //hdl/ip/vhd/axi_blocks:axil_interconnect_tb`
+
+The harness drives the flat-port entity with a mixed responder map (pipe depths 0, 1, 3 and 2, plus
+a deliberately unmapped hole) and can be driven either by `vunit_lib.axi_lite_master` or by hand, so
+the testbench can reproduce initiator handshake patterns the bus functional model never generates.
+
+Two responder models with deliberately different handshake shapes:
+
+* `axil_sram_responder` wraps the production `axil_target_txn`, so it reproduces the contract every
+ register block in the tree presents.
+* `axil_slow_responder` accepts AW and W independently after LFSR-driven stalls and holds its
+ responses until READY — a shape nothing in the tree currently exercises.
+
+The harness also counts handshakes on both sides of the fabric, so a duplicated transaction is
+caught even when the data happens to land correctly, and checks that a stalled channel's payload
+stays put.
+
+[NOTE]
+====
+`write_axi_lite` only *queues* a write; it does not block. Call `wait_until_idle` before taking the
+bus over by hand or before sampling the handshake counters.
+====
+
+== Things worth knowing
+
+* A concurrent read and write from one initiator is serialized, not run in parallel. Neither
+ initiator does this today.
+* `axil_interconnect` hard-codes a 26-bit initiator. A differently sized initiator needs either the
+ 2008 entity directly (as cosmo_hp does) or a new wrapper.
+* Registering B and R at the *initiator* boundary would cut the remaining combinational return mux,
+ but `RDATA` cannot be registered without also delaying `RVALID`, which would cost a cycle for
+ every responder including the unpiped ones. If it is ever needed it belongs behind its own
+ generic, not `pipe_stages`.
diff --git a/hdl/ip/vhd/axi_blocks/docs/axil_interconnect_block.drawio.svg b/hdl/ip/vhd/axi_blocks/docs/axil_interconnect_block.drawio.svg
new file mode 100644
index 00000000..efdbf957
--- /dev/null
+++ b/hdl/ip/vhd/axi_blocks/docs/axil_interconnect_block.drawio.svg
@@ -0,0 +1,83 @@
+
+
+
+
diff --git a/hdl/ip/vhd/axi_blocks/docs/axil_pipe.drawio.svg b/hdl/ip/vhd/axi_blocks/docs/axil_pipe.drawio.svg
new file mode 100644
index 00000000..11a20e41
--- /dev/null
+++ b/hdl/ip/vhd/axi_blocks/docs/axil_pipe.drawio.svg
@@ -0,0 +1,83 @@
+
+
+
+
diff --git a/hdl/ip/vhd/axi_blocks/sims/axil_interconnect_sim_pkg.vhd b/hdl/ip/vhd/axi_blocks/sims/axil_interconnect_sim_pkg.vhd
new file mode 100644
index 00000000..3ecce019
--- /dev/null
+++ b/hdl/ip/vhd/axi_blocks/sims/axil_interconnect_sim_pkg.vhd
@@ -0,0 +1,76 @@
+-- This Source Code Form is subject to the terms of the Mozilla Public
+-- License, v. 2.0. If a copy of the MPL was not distributed with this
+-- file, You can obtain one at https://mozilla.org/MPL/2.0/.
+
+library ieee;
+use ieee.std_logic_1164.all;
+use ieee.numeric_std.all;
+
+library vunit_lib;
+ context vunit_lib.vunit_context;
+ context vunit_lib.com_context;
+ context vunit_lib.vc_context;
+
+use work.axil_common_pkg.all;
+
+package axil_interconnect_sim_pkg is
+
+ -- The fabric under test is instantiated with a 26 bit initiator, matching the
+ -- FMC target in the real designs.
+ constant INITIATOR_ADDR_WIDTH : integer := 26;
+
+ constant bus_handle : bus_master_t := new_bus(
+ data_length => 32,
+ address_length => INITIATOR_ADDR_WIDTH
+ );
+
+ -- Responder map for the harness. The gap between 0x300 and 0x7FFF, plus
+ -- everything at or above 0x10000, is deliberately unmapped so the catch-all
+ -- SLVERR path gets exercised.
+ constant SRAM_A_IDX : integer := 0;
+ constant SRAM_B_IDX : integer := 1;
+ constant SLOW_IDX : integer := 2;
+ constant WIDE_IDX : integer := 3;
+
+ -- Drives both the deepest pipe in the map below and the slow responder's worst
+ -- case stall, so one number sets how much delay the testbench has to tolerate
+ -- on either side of the fabric.
+ constant MAX_DELAY : integer := 3;
+
+ constant config_array : axil_responder_cfg_array_t(0 to 3) :=
+ (SRAM_A_IDX => resp_cfg(base_addr => x"00000000", addr_span_bits => 8, pipe_stages => 0),
+ SRAM_B_IDX => resp_cfg(base_addr => x"00000100", addr_span_bits => 8, pipe_stages => 1),
+ SLOW_IDX => resp_cfg(base_addr => x"00000200", addr_span_bits => 8, pipe_stages => MAX_DELAY),
+ WIDE_IDX => resp_cfg(base_addr => x"00008000", addr_span_bits => 15, pipe_stages => 2));
+
+ -- An integer as an initiator-width bus address
+ function bus_addr (constant addr : integer) return std_logic_vector;
+
+ -- Bus address of a register in a responder: its configured base address
+ -- plus a byte offset
+ function responder_addr (constant idx : integer; constant offset : integer) return std_logic_vector;
+
+ -- An integer as a 32 bit AXI data word
+ function data_word (constant value : integer) return std_logic_vector;
+
+end package;
+
+package body axil_interconnect_sim_pkg is
+
+ function bus_addr (constant addr : integer) return std_logic_vector is
+ begin
+ return std_logic_vector(to_unsigned(addr, INITIATOR_ADDR_WIDTH));
+ end function;
+
+ function responder_addr (constant idx : integer; constant offset : integer) return std_logic_vector is
+ begin
+ return std_logic_vector(unsigned(config_array(idx).base_addr(INITIATOR_ADDR_WIDTH - 1 downto 0))
+ + to_unsigned(offset, INITIATOR_ADDR_WIDTH));
+ end function;
+
+ function data_word (constant value : integer) return std_logic_vector is
+ begin
+ return std_logic_vector(to_unsigned(value, 32));
+ end function;
+
+end package body;
diff --git a/hdl/ip/vhd/axi_blocks/sims/axil_interconnect_tb.vhd b/hdl/ip/vhd/axi_blocks/sims/axil_interconnect_tb.vhd
new file mode 100644
index 00000000..0d80c750
--- /dev/null
+++ b/hdl/ip/vhd/axi_blocks/sims/axil_interconnect_tb.vhd
@@ -0,0 +1,403 @@
+-- This Source Code Form is subject to the terms of the Mozilla Public
+-- License, v. 2.0. If a copy of the MPL was not distributed with this
+-- file, You can obtain one at https://mozilla.org/MPL/2.0/.
+
+library ieee;
+use ieee.std_logic_1164.all;
+use ieee.numeric_std.all;
+
+library vunit_lib;
+ context vunit_lib.com_context;
+ context vunit_lib.vunit_context;
+ context vunit_lib.vc_context;
+use vunit_lib.axi_lite_master_pkg.all;
+
+use work.axil_common_pkg.all;
+use work.axil_interconnect_sim_pkg.all;
+
+entity axil_interconnect_tb is
+ generic (
+ runner_cfg : string
+ );
+end entity;
+
+architecture tb of axil_interconnect_tb is
+
+begin
+
+ th: entity work.axil_interconnect_th;
+
+ -- Note: external names are broken in the GHDL llvm backend
+ -- (https://github.com/ghdl/ghdl/issues/2610) so this sim is nvc only.
+ bench: process
+ alias clk is <>;
+ alias reset is <>;
+ alias reset_force is <>;
+
+ alias man_mode is <>;
+ alias man_awvalid is <>;
+ alias man_awaddr is
+ <>;
+ alias man_wvalid is <>;
+ alias man_wdata is <>;
+ alias man_wstrb is <>;
+ alias man_bready is <>;
+ alias man_arvalid is <>;
+ alias man_araddr is
+ <>;
+ alias man_rready is <>;
+
+ alias init_awready is <>;
+ alias init_bvalid is <>;
+ alias init_bresp is <>;
+ alias init_rvalid is <>;
+ alias init_rdata is <>;
+ alias init_rresp is <>;
+
+ alias init_aw_hs is <>;
+ alias init_b_hs is <>;
+ alias init_ar_hs is <>;
+ alias init_r_hs is <>;
+ alias resp_aw_hs is <>;
+ alias resp_ar_hs is <>;
+
+ -- generous enough that a working fabric never hits it, short enough that
+ -- a wedged one fails instead of running the watchdog out
+ constant TXN_TIMEOUT : time := 5 us;
+
+ variable rdata : std_logic_vector(31 downto 0);
+ variable lat0 : integer;
+ variable lat1 : integer;
+ variable lat2 : integer;
+
+ -- Every accepted write must produce exactly one B, every accepted read
+ -- exactly one R, and (for mapped addresses) exactly one responder side
+ -- address handshake. A duplicated write shows up as resp_aw_hs running
+ -- ahead of init_aw_hs even when the data lands correctly.
+ procedure check_handshake_accounting (
+ signal net : inout network_t;
+ constant mapped_only : boolean
+ ) is
+ begin
+ -- writes are only queued by write_axi_lite, so let the bus
+ -- functional model drain before counting anything
+ wait_until_idle(net, bus_handle);
+ check_equal(init_aw_hs, init_b_hs, "write address and write response handshakes disagree");
+ check_equal(init_ar_hs, init_r_hs, "read address and read data handshakes disagree");
+ if mapped_only then
+ check_equal(resp_aw_hs, init_aw_hs,
+ "responder saw a different number of writes than the initiator issued");
+ check_equal(resp_ar_hs, init_ar_hs,
+ "responder saw a different number of reads than the initiator issued");
+ end if;
+ end procedure;
+
+ procedure clear_manual is
+ begin
+ man_awvalid <= '0';
+ man_wvalid <= '0';
+ man_bready <= '0';
+ man_arvalid <= '0';
+ man_rready <= '0';
+ wait until rising_edge(clk);
+ man_mode <= '0';
+ wait until rising_edge(clk);
+ end procedure;
+
+ -- Drive one write straight at the fabric, emulating the FMC target:
+ -- AWVALID goes up first and WVALID only follows once the write data FIFO
+ -- has something in it.
+ procedure manual_write (
+ constant addr : in std_logic_vector;
+ constant data : in std_logic_vector(31 downto 0);
+ constant aw_lead : in integer;
+ constant expect_resp : in std_logic_vector(1 downto 0)
+ ) is
+ begin
+ man_mode <= '1';
+ man_awaddr <= addr;
+ man_wdata <= data;
+ man_wstrb <= "1111";
+ man_bready <= '1';
+ man_awvalid <= '1';
+ man_wvalid <= '0';
+
+ -- While the initiator has only presented AW, the fabric must not
+ -- accept the write. If it does, the FMC clears AWVALID early and
+ -- then re-raises it, which is what wedges the bus.
+ for i in 1 to aw_lead loop
+ wait until rising_edge(clk);
+ check_equal(init_awready, '0', "fabric asserted AWREADY before WVALID was presented");
+ end loop;
+
+ man_wvalid <= '1';
+ wait until rising_edge(clk) and init_bvalid = '1' for TXN_TIMEOUT;
+ check_equal(init_bvalid, '1', "timed out waiting for the write response");
+ check_equal(init_bresp, expect_resp, "unexpected write response");
+ wait until rising_edge(clk);
+ man_awvalid <= '0';
+ man_wvalid <= '0';
+ wait until rising_edge(clk);
+ end procedure;
+
+ procedure manual_read (
+ constant addr : in std_logic_vector;
+ variable data : out std_logic_vector(31 downto 0);
+ constant expect_resp : in std_logic_vector(1 downto 0)
+ ) is
+ begin
+ man_mode <= '1';
+ man_araddr <= addr;
+ man_rready <= '1';
+ man_arvalid <= '1';
+ wait until rising_edge(clk) and init_rvalid = '1' for TXN_TIMEOUT;
+ check_equal(init_rvalid, '1', "timed out waiting for read data");
+ data := init_rdata;
+ check_equal(init_rresp, expect_resp, "unexpected read response");
+ wait until rising_edge(clk);
+ man_arvalid <= '0';
+ man_rready <= '0';
+ wait until rising_edge(clk);
+ end procedure;
+
+ -- Like manual_read, but also reports how many clocks the fabric took to
+ -- answer, so the testbench can show the configured stages are really in
+ -- the path rather than being generated away.
+ procedure manual_read_timed (
+ constant addr : in std_logic_vector;
+ variable data : out std_logic_vector(31 downto 0);
+ variable cycles : out integer
+ ) is
+ variable count : integer := 0;
+ begin
+ man_mode <= '1';
+ man_araddr <= addr;
+ man_rready <= '1';
+ man_arvalid <= '1';
+ loop
+ wait until rising_edge(clk);
+ exit when init_rvalid = '1';
+ count := count + 1;
+ if count > 100 then
+ check(false, "timed out waiting for read data");
+ exit;
+ end if;
+ end loop;
+ data := init_rdata;
+ cycles := count;
+ man_arvalid <= '0';
+ man_rready <= '0';
+ wait until rising_edge(clk);
+ end procedure;
+
+ begin
+ test_runner_setup(runner, runner_cfg);
+ wait until reset = '0';
+ wait for 500 ns;
+
+ while test_suite loop
+ if run("write_read_each_responder") then
+ for idx in config_array'range loop
+ write_axi_lite(net, bus_handle, responder_addr(idx, 16#00#), x"C0DE0000" or data_word(idx * 16));
+ write_axi_lite(net, bus_handle, responder_addr(idx, 16#08#), x"FEED0000" or data_word(idx * 16));
+ end loop;
+ -- read back after all the writes, so a fabric that leaks a write
+ -- into the wrong responder is caught rather than masked
+ for idx in config_array'range loop
+ check_axi_lite(net, bus_handle, responder_addr(idx, 16#00#), axi_resp_okay,
+ x"C0DE0000" or data_word(idx * 16), "responder 0x00 readback");
+ check_axi_lite(net, bus_handle, responder_addr(idx, 16#08#), axi_resp_okay,
+ x"FEED0000" or data_word(idx * 16), "responder 0x08 readback");
+ end loop;
+ check_handshake_accounting(net, mapped_only => true);
+
+ elsif run("back_to_back_same_responder") then
+ -- run the burst against every responder in turn, so token and
+ -- payload reuse is covered at each configured pipe depth
+ for idx in config_array'range loop
+ for word in 0 to 7 loop
+ write_axi_lite(net, bus_handle, responder_addr(idx, 4 * word),
+ x"A5A50000" or data_word(16 * idx + word));
+ end loop;
+ for word in 0 to 7 loop
+ check_axi_lite(net, bus_handle, responder_addr(idx, 4 * word), axi_resp_okay,
+ x"A5A50000" or data_word(16 * idx + word), "back to back readback");
+ end loop;
+ end loop;
+ check_handshake_accounting(net, mapped_only => true);
+
+ elsif run("back_to_back_alternating") then
+ -- alternate between the fastest and the slowest responder with no
+ -- idle time in between
+ for word in 0 to 7 loop
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 4 * word),
+ x"11110000" or data_word(word));
+ write_axi_lite(net, bus_handle, responder_addr(SLOW_IDX, 4 * word), x"22220000" or data_word(word));
+ end loop;
+ for word in 0 to 7 loop
+ check_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 4 * word), axi_resp_okay,
+ x"11110000" or data_word(word), "sram_a readback");
+ check_axi_lite(net, bus_handle, responder_addr(SLOW_IDX, 4 * word), axi_resp_okay,
+ x"22220000" or data_word(word), "slow readback");
+ end loop;
+ check_handshake_accounting(net, mapped_only => true);
+
+ elsif run("read_after_write_same_addr") then
+ for idx in config_array'range loop
+ write_axi_lite(net, bus_handle, responder_addr(idx, 16#10#), x"5A5A1234");
+ check_axi_lite(net, bus_handle, responder_addr(idx, 16#10#), axi_resp_okay, x"5A5A1234",
+ "read immediately after write");
+ write_axi_lite(net, bus_handle, responder_addr(idx, 16#10#), x"A5A54321");
+ check_axi_lite(net, bus_handle, responder_addr(idx, 16#10#), axi_resp_okay, x"A5A54321",
+ "read immediately after overwrite");
+ end loop;
+ check_handshake_accounting(net, mapped_only => true);
+
+ elsif run("unmapped_slverr") then
+ -- the gap between the 8 bit responders and the wide one, the top
+ -- of that gap, and an address past every responder
+ write_axi_lite(net, bus_handle, bus_addr(16#000300#), x"DEADDEAD", axi_resp_slverr);
+ check_axi_lite(net, bus_handle, bus_addr(16#000300#), axi_resp_slverr, x"DEADBEEF",
+ "unmapped read at 0x300");
+ write_axi_lite(net, bus_handle, bus_addr(16#007FFC#), x"DEADDEAD", axi_resp_slverr);
+ check_axi_lite(net, bus_handle, bus_addr(16#007FFC#), axi_resp_slverr, x"DEADBEEF",
+ "unmapped read at 0x7FFC");
+ write_axi_lite(net, bus_handle, bus_addr(16#010000#), x"DEADDEAD", axi_resp_slverr);
+ check_axi_lite(net, bus_handle, bus_addr(16#010000#), axi_resp_slverr, x"DEADBEEF",
+ "unmapped read at 0x10000");
+ -- an unmapped access must not have disturbed a mapped one
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#04#), x"600D600D");
+ check_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#04#), axi_resp_okay, x"600D600D",
+ "mapped access after unmapped");
+ check_handshake_accounting(net, mapped_only => false);
+
+ elsif run("boundary_addresses") then
+ -- first and last word of each mapped region, which is where an
+ -- equality based decode and a magnitude compare could disagree
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#00#), x"00000001");
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#FC#), x"000000FC");
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_B_IDX, 16#00#), x"00000100");
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_B_IDX, 16#FC#), x"000001FC");
+ write_axi_lite(net, bus_handle, responder_addr(WIDE_IDX, 16#0000#), x"00008000");
+ write_axi_lite(net, bus_handle, responder_addr(WIDE_IDX, 16#0FFC#), x"00008FFC");
+
+ -- 0x00 and 0xFC land in different words of sram_a, and sram_b's
+ -- 0x100 must not have aliased on top of sram_a's 0x00
+ check_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#00#), axi_resp_okay, x"00000001",
+ "sram_a low boundary");
+ check_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#FC#), axi_resp_okay, x"000000FC",
+ "sram_a high boundary");
+ check_axi_lite(net, bus_handle, responder_addr(SRAM_B_IDX, 16#00#), axi_resp_okay, x"00000100",
+ "sram_b low boundary");
+ check_axi_lite(net, bus_handle, responder_addr(SRAM_B_IDX, 16#FC#), axi_resp_okay, x"000001FC",
+ "sram_b high boundary");
+ check_axi_lite(net, bus_handle, responder_addr(WIDE_IDX, 16#0000#), axi_resp_okay, x"00008000",
+ "wide low boundary");
+ check_axi_lite(net, bus_handle, responder_addr(WIDE_IDX, 16#0FFC#), axi_resp_okay, x"00008FFC",
+ "wide high boundary");
+ check_handshake_accounting(net, mapped_only => true);
+
+ elsif run("glitchy_aw_initiator") then
+ manual_write(responder_addr(SRAM_A_IDX, 16#10#), x"6060BEEF", 6, OKAY);
+ clear_manual;
+ check_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#10#), axi_resp_okay, x"6060BEEF",
+ "readback after an AW leading write");
+ check_handshake_accounting(net, mapped_only => true);
+
+ elsif run("glitchy_aw_unmapped") then
+ -- the catch-all responder is the one that used to assert AWREADY
+ -- with no regard for WVALID
+ manual_write(bus_addr(16#000300#), x"6060BEEF", 6, SLVERR);
+ clear_manual;
+ check_axi_lite(net, bus_handle, bus_addr(16#000300#), axi_resp_slverr, x"DEADBEEF",
+ "unmapped read after an AW leading write");
+ check_handshake_accounting(net, mapped_only => false);
+
+ elsif run("stuck_awvalid") then
+ -- seed a known value in sram_a that the stuck write must not
+ -- be able to shadow. write_axi_lite only queues the write, so
+ -- wait for the bus functional model to actually retire it
+ -- before taking the bus over by hand.
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#04#), x"600DF00D");
+ wait_until_idle(net, bus_handle);
+
+ man_mode <= '1';
+ man_awaddr <= responder_addr(SRAM_B_IDX, 16#00#);
+ man_wdata <= x"11112222";
+ man_wstrb <= "1111";
+ man_bready <= '1';
+ man_awvalid <= '1';
+ man_wvalid <= '1';
+ wait until rising_edge(clk) and init_bvalid = '1' for TXN_TIMEOUT;
+ check_equal(init_bvalid, '1', "timed out waiting for the write response");
+ wait until rising_edge(clk);
+
+ -- the FMC only clears AWVALID on an AWREADY it happened to
+ -- observe, so it can still be asserted here. The fabric has to
+ -- tear the transaction down anyway and decode the next one.
+ man_wvalid <= '0';
+ man_bready <= '0';
+ manual_read(responder_addr(SRAM_A_IDX, 16#04#), rdata, OKAY);
+ check_equal(rdata, std_logic_vector'(x"600DF00D"),
+ "read decoded against a stale write address");
+ clear_manual;
+ check_handshake_accounting(net, mapped_only => true);
+
+ elsif run("pipe_latency") then
+ -- sram_a has no pipe, sram_b has one stage and the wide responder
+ -- two, so the answer must arrive strictly later each time
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#00#), x"00000011");
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_B_IDX, 16#00#), x"00000022");
+ write_axi_lite(net, bus_handle, responder_addr(WIDE_IDX, 16#00#), x"00000033");
+ wait_until_idle(net, bus_handle);
+
+ manual_read_timed(responder_addr(SRAM_A_IDX, 16#00#), rdata, lat0);
+ check_equal(rdata, std_logic_vector'(x"00000011"), "unpiped readback");
+ manual_read_timed(responder_addr(SRAM_B_IDX, 16#00#), rdata, lat1);
+ check_equal(rdata, std_logic_vector'(x"00000022"), "one stage readback");
+ manual_read_timed(responder_addr(WIDE_IDX, 16#00#), rdata, lat2);
+ check_equal(rdata, std_logic_vector'(x"00000033"), "two stage readback");
+ info("read latency in clocks: 0 stages=" & to_string(lat0) &
+ " 1 stage=" & to_string(lat1) & " 2 stages=" & to_string(lat2));
+ check(lat1 > lat0, "the one stage pipe added no latency");
+ check(lat2 > lat1, "the two stage pipe added no latency over one stage");
+ clear_manual;
+ check_handshake_accounting(net, mapped_only => true);
+
+ elsif run("reset_mid_transaction") then
+ -- kick off a read at the slow responder and reset while it is in
+ -- flight, then confirm the fabric comes back clean
+ man_mode <= '1';
+ man_araddr <= responder_addr(SLOW_IDX, 16#00#);
+ man_rready <= '1';
+ man_arvalid <= '1';
+ wait for 40 ns;
+ reset_force <= '1';
+ wait for 40 ns;
+ man_arvalid <= '0';
+ man_rready <= '0';
+ wait for 40 ns;
+ reset_force <= '0';
+ wait until reset = '0';
+ wait for 200 ns;
+ man_mode <= '0';
+ wait for 200 ns;
+
+ write_axi_lite(net, bus_handle, responder_addr(SLOW_IDX, 16#00#), x"AF7E8E5E");
+ check_axi_lite(net, bus_handle, responder_addr(SLOW_IDX, 16#00#), axi_resp_okay, x"AF7E8E5E",
+ "slow responder after a mid transaction reset");
+ write_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#00#), x"C1EA4000");
+ check_axi_lite(net, bus_handle, responder_addr(SRAM_A_IDX, 16#00#), axi_resp_okay, x"C1EA4000",
+ "sram_a after a mid transaction reset");
+ check_handshake_accounting(net, mapped_only => true);
+ end if;
+ end loop;
+
+ wait for 2 us;
+ test_runner_cleanup(runner);
+ wait;
+ end process;
+
+ test_runner_watchdog(runner, 10 ms);
+
+end tb;
diff --git a/hdl/ip/vhd/axi_blocks/sims/axil_interconnect_th.vhd b/hdl/ip/vhd/axi_blocks/sims/axil_interconnect_th.vhd
new file mode 100644
index 00000000..379e86d2
--- /dev/null
+++ b/hdl/ip/vhd/axi_blocks/sims/axil_interconnect_th.vhd
@@ -0,0 +1,362 @@
+-- This Source Code Form is subject to the terms of the Mozilla Public
+-- License, v. 2.0. If a copy of the MPL was not distributed with this
+-- file, You can obtain one at https://mozilla.org/MPL/2.0/.
+
+library ieee;
+use ieee.std_logic_1164.all;
+use ieee.numeric_std.all;
+
+library vunit_lib;
+ context vunit_lib.vunit_context;
+ context vunit_lib.com_context;
+ context vunit_lib.vc_context;
+
+use work.axil_common_pkg.all;
+use work.axilite_if_2k8_pkg.all;
+use work.axil_interconnect_sim_pkg.all;
+
+entity axil_interconnect_th is
+end entity;
+
+architecture th of axil_interconnect_th is
+
+ signal clk : std_logic := '0';
+ signal reset : std_logic;
+ signal reset_por : std_logic := '1';
+ -- testbench driven, so a reset can be injected mid transaction
+ signal reset_force : std_logic := '0';
+
+ -- The bus functional model drives these
+ signal bfm_awvalid : std_logic;
+ signal bfm_awaddr : std_logic_vector(INITIATOR_ADDR_WIDTH - 1 downto 0);
+ signal bfm_wvalid : std_logic;
+ signal bfm_wdata : std_logic_vector(31 downto 0);
+ signal bfm_wstrb : std_logic_vector(3 downto 0);
+ signal bfm_bready : std_logic;
+ signal bfm_arvalid : std_logic;
+ signal bfm_araddr : std_logic_vector(INITIATOR_ADDR_WIDTH - 1 downto 0);
+ signal bfm_rready : std_logic;
+
+ -- The testbench drives these directly when man_mode is set, so it can
+ -- reproduce initiator handshake patterns the BFM never generates (notably
+ -- the FMC target's habit of raising AWVALID well before WVALID, and
+ -- re-raising it after an AWREADY pulse).
+ signal man_mode : std_logic := '0';
+ signal man_awvalid : std_logic := '0';
+ signal man_awaddr : std_logic_vector(INITIATOR_ADDR_WIDTH - 1 downto 0) := (others => '0');
+ signal man_wvalid : std_logic := '0';
+ signal man_wdata : std_logic_vector(31 downto 0) := (others => '0');
+ signal man_wstrb : std_logic_vector(3 downto 0) := "1111";
+ signal man_bready : std_logic := '0';
+ signal man_arvalid : std_logic := '0';
+ signal man_araddr : std_logic_vector(INITIATOR_ADDR_WIDTH - 1 downto 0) := (others => '0');
+ signal man_rready : std_logic := '0';
+
+ -- initiator side of the fabric
+ signal init_awvalid : std_logic;
+ signal init_awready : std_logic;
+ signal init_awaddr : std_logic_vector(INITIATOR_ADDR_WIDTH - 1 downto 0);
+ signal init_wvalid : std_logic;
+ signal init_wready : std_logic;
+ signal init_wdata : std_logic_vector(31 downto 0);
+ signal init_wstrb : std_logic_vector(3 downto 0);
+ signal init_bvalid : std_logic;
+ signal init_bready : std_logic;
+ signal init_bresp : std_logic_vector(1 downto 0);
+ signal init_arvalid : std_logic;
+ signal init_arready : std_logic;
+ signal init_araddr : std_logic_vector(INITIATOR_ADDR_WIDTH - 1 downto 0);
+ signal init_rvalid : std_logic;
+ signal init_rready : std_logic;
+ signal init_rdata : std_logic_vector(31 downto 0);
+ signal init_rresp : std_logic_vector(1 downto 0);
+
+ -- responder side of the fabric
+ signal responders_write_address_valid : std_logic_vector(config_array'range);
+ signal responders_write_address_ready : std_logic_vector(config_array'range);
+ signal responders_write_address_addr : tgt_addr32_t(config_array'range);
+ signal responders_write_data_valid : std_logic_vector(config_array'range);
+ signal responders_write_data_ready : std_logic_vector(config_array'range);
+ signal responders_write_data_data : tgt_dat32_t(config_array'range);
+ signal responders_write_data_strb : tgt_strb_t(config_array'range);
+ signal responders_write_response_ready : std_logic_vector(config_array'range);
+ signal responders_write_response_resp : tgt_resp_t(config_array'range);
+ signal responders_write_response_valid : std_logic_vector(config_array'range);
+ signal responders_read_address_valid : std_logic_vector(config_array'range);
+ signal responders_read_address_addr : tgt_addr32_t(config_array'range);
+ signal responders_read_address_ready : std_logic_vector(config_array'range);
+ signal responders_read_data_ready : std_logic_vector(config_array'range);
+ signal responders_read_data_resp : tgt_resp_t(config_array'range);
+ signal responders_read_data_valid : std_logic_vector(config_array'range);
+ signal responders_read_data_data : tgt_dat32_t(config_array'range);
+
+ -- handshake accounting, checked by the testbench
+ signal init_aw_hs : integer := 0;
+ signal init_b_hs : integer := 0;
+ signal init_ar_hs : integer := 0;
+ signal init_r_hs : integer := 0;
+ signal resp_aw_hs : integer := 0;
+ signal resp_ar_hs : integer := 0;
+
+begin
+
+ -- 125 MHz, matching the fabric clock in cosmo_seq and grapefruit
+ clk <= not clk after 4 ns;
+ reset_por <= '0' after 200 ns;
+ reset <= reset_por or reset_force;
+
+ axi_lite_master_inst: entity vunit_lib.axi_lite_master
+ generic map (
+ bus_handle => bus_handle
+ )
+ port map (
+ aclk => clk,
+ arready => init_arready,
+ arvalid => bfm_arvalid,
+ araddr => bfm_araddr,
+ rready => bfm_rready,
+ rvalid => init_rvalid,
+ rdata => init_rdata,
+ rresp => init_rresp,
+ awready => init_awready,
+ awvalid => bfm_awvalid,
+ awaddr => bfm_awaddr,
+ wready => init_wready,
+ wvalid => bfm_wvalid,
+ wdata => bfm_wdata,
+ wstrb => bfm_wstrb,
+ bvalid => init_bvalid,
+ bready => bfm_bready,
+ bresp => init_bresp
+ );
+
+ init_awvalid <= man_awvalid when man_mode = '1' else bfm_awvalid;
+ init_awaddr <= man_awaddr when man_mode = '1' else bfm_awaddr;
+ init_wvalid <= man_wvalid when man_mode = '1' else bfm_wvalid;
+ init_wdata <= man_wdata when man_mode = '1' else bfm_wdata;
+ init_wstrb <= man_wstrb when man_mode = '1' else bfm_wstrb;
+ init_bready <= man_bready when man_mode = '1' else bfm_bready;
+ init_arvalid <= man_arvalid when man_mode = '1' else bfm_arvalid;
+ init_araddr <= man_araddr when man_mode = '1' else bfm_araddr;
+ init_rready <= man_rready when man_mode = '1' else bfm_rready;
+
+ dut: entity work.axil_interconnect_2k8
+ generic map (
+ initiator_addr_width => INITIATOR_ADDR_WIDTH,
+ config_array => config_array
+ )
+ port map (
+ clk => clk,
+ reset => reset,
+ initiator_write_address_addr => init_awaddr,
+ initiator_write_address_valid => init_awvalid,
+ initiator_write_address_ready => init_awready,
+ initiator_write_data_data => init_wdata,
+ initiator_write_data_strb => init_wstrb,
+ initiator_write_data_ready => init_wready,
+ initiator_write_data_valid => init_wvalid,
+ initiator_write_response_valid => init_bvalid,
+ initiator_write_response_resp => init_bresp,
+ initiator_write_response_ready => init_bready,
+ initiator_read_address_addr => init_araddr,
+ initiator_read_address_ready => init_arready,
+ initiator_read_address_valid => init_arvalid,
+ initiator_read_data_valid => init_rvalid,
+ initiator_read_data_ready => init_rready,
+ initiator_read_data_resp => init_rresp,
+ initiator_read_data_data => init_rdata,
+ responders_write_address_valid => responders_write_address_valid,
+ responders_write_address_ready => responders_write_address_ready,
+ responders_write_address_addr => responders_write_address_addr,
+ responders_write_data_valid => responders_write_data_valid,
+ responders_write_data_ready => responders_write_data_ready,
+ responders_write_data_data => responders_write_data_data,
+ responders_write_data_strb => responders_write_data_strb,
+ responders_write_response_ready => responders_write_response_ready,
+ responders_read_address_valid => responders_read_address_valid,
+ responders_read_address_addr => responders_read_address_addr,
+ responders_read_data_ready => responders_read_data_ready,
+ responders_write_response_resp => responders_write_response_resp,
+ responders_write_response_valid => responders_write_response_valid,
+ responders_read_address_ready => responders_read_address_ready,
+ responders_read_data_resp => responders_read_data_resp,
+ responders_read_data_valid => responders_read_data_valid,
+ responders_read_data_data => responders_read_data_data
+ );
+
+ sram_a: entity work.axil_sram_responder
+ generic map (
+ addr_width => 8
+ )
+ port map (
+ clk => clk,
+ reset => reset,
+ awvalid => responders_write_address_valid(SRAM_A_IDX),
+ awready => responders_write_address_ready(SRAM_A_IDX),
+ awaddr => responders_write_address_addr(SRAM_A_IDX)(7 downto 0),
+ wvalid => responders_write_data_valid(SRAM_A_IDX),
+ wready => responders_write_data_ready(SRAM_A_IDX),
+ wdata => responders_write_data_data(SRAM_A_IDX),
+ wstrb => responders_write_data_strb(SRAM_A_IDX),
+ bvalid => responders_write_response_valid(SRAM_A_IDX),
+ bready => responders_write_response_ready(SRAM_A_IDX),
+ bresp => responders_write_response_resp(SRAM_A_IDX),
+ arvalid => responders_read_address_valid(SRAM_A_IDX),
+ arready => responders_read_address_ready(SRAM_A_IDX),
+ araddr => responders_read_address_addr(SRAM_A_IDX)(7 downto 0),
+ rvalid => responders_read_data_valid(SRAM_A_IDX),
+ rready => responders_read_data_ready(SRAM_A_IDX),
+ rdata => responders_read_data_data(SRAM_A_IDX),
+ rresp => responders_read_data_resp(SRAM_A_IDX)
+ );
+
+ sram_b: entity work.axil_sram_responder
+ generic map (
+ addr_width => 8
+ )
+ port map (
+ clk => clk,
+ reset => reset,
+ awvalid => responders_write_address_valid(SRAM_B_IDX),
+ awready => responders_write_address_ready(SRAM_B_IDX),
+ awaddr => responders_write_address_addr(SRAM_B_IDX)(7 downto 0),
+ wvalid => responders_write_data_valid(SRAM_B_IDX),
+ wready => responders_write_data_ready(SRAM_B_IDX),
+ wdata => responders_write_data_data(SRAM_B_IDX),
+ wstrb => responders_write_data_strb(SRAM_B_IDX),
+ bvalid => responders_write_response_valid(SRAM_B_IDX),
+ bready => responders_write_response_ready(SRAM_B_IDX),
+ bresp => responders_write_response_resp(SRAM_B_IDX),
+ arvalid => responders_read_address_valid(SRAM_B_IDX),
+ arready => responders_read_address_ready(SRAM_B_IDX),
+ araddr => responders_read_address_addr(SRAM_B_IDX)(7 downto 0),
+ rvalid => responders_read_data_valid(SRAM_B_IDX),
+ rready => responders_read_data_ready(SRAM_B_IDX),
+ rdata => responders_read_data_data(SRAM_B_IDX),
+ rresp => responders_read_data_resp(SRAM_B_IDX)
+ );
+
+ slow: entity work.axil_slow_responder
+ generic map (
+ addr_width => 8,
+ max_delay => MAX_DELAY,
+ seed => x"5A"
+ )
+ port map (
+ clk => clk,
+ reset => reset,
+ awvalid => responders_write_address_valid(SLOW_IDX),
+ awready => responders_write_address_ready(SLOW_IDX),
+ awaddr => responders_write_address_addr(SLOW_IDX)(7 downto 0),
+ wvalid => responders_write_data_valid(SLOW_IDX),
+ wready => responders_write_data_ready(SLOW_IDX),
+ wdata => responders_write_data_data(SLOW_IDX),
+ wstrb => responders_write_data_strb(SLOW_IDX),
+ bvalid => responders_write_response_valid(SLOW_IDX),
+ bready => responders_write_response_ready(SLOW_IDX),
+ bresp => responders_write_response_resp(SLOW_IDX),
+ arvalid => responders_read_address_valid(SLOW_IDX),
+ arready => responders_read_address_ready(SLOW_IDX),
+ araddr => responders_read_address_addr(SLOW_IDX)(7 downto 0),
+ rvalid => responders_read_data_valid(SLOW_IDX),
+ rready => responders_read_data_ready(SLOW_IDX),
+ rdata => responders_read_data_data(SLOW_IDX),
+ rresp => responders_read_data_resp(SLOW_IDX)
+ );
+
+ wide: entity work.axil_sram_responder
+ generic map (
+ addr_width => 15
+ )
+ port map (
+ clk => clk,
+ reset => reset,
+ awvalid => responders_write_address_valid(WIDE_IDX),
+ awready => responders_write_address_ready(WIDE_IDX),
+ awaddr => responders_write_address_addr(WIDE_IDX)(14 downto 0),
+ wvalid => responders_write_data_valid(WIDE_IDX),
+ wready => responders_write_data_ready(WIDE_IDX),
+ wdata => responders_write_data_data(WIDE_IDX),
+ wstrb => responders_write_data_strb(WIDE_IDX),
+ bvalid => responders_write_response_valid(WIDE_IDX),
+ bready => responders_write_response_ready(WIDE_IDX),
+ bresp => responders_write_response_resp(WIDE_IDX),
+ arvalid => responders_read_address_valid(WIDE_IDX),
+ arready => responders_read_address_ready(WIDE_IDX),
+ araddr => responders_read_address_addr(WIDE_IDX)(14 downto 0),
+ rvalid => responders_read_data_valid(WIDE_IDX),
+ rready => responders_read_data_ready(WIDE_IDX),
+ rdata => responders_read_data_data(WIDE_IDX),
+ rresp => responders_read_data_resp(WIDE_IDX)
+ );
+
+ -- Handshake accounting. The testbench compares these at the end of every
+ -- test: one initiator write must produce exactly one responder-side AW
+ -- handshake, so a fabric or pipeline stage that issues a duplicate write
+ -- shows up here even when the data happens to land correctly.
+ counters: process(clk, reset)
+ begin
+ if reset = '1' then
+ init_aw_hs <= 0;
+ init_b_hs <= 0;
+ init_ar_hs <= 0;
+ init_r_hs <= 0;
+ resp_aw_hs <= 0;
+ resp_ar_hs <= 0;
+ elsif rising_edge(clk) then
+ if init_awvalid = '1' and init_awready = '1' then
+ init_aw_hs <= init_aw_hs + 1;
+ end if;
+ if init_bvalid = '1' and init_bready = '1' then
+ init_b_hs <= init_b_hs + 1;
+ end if;
+ if init_arvalid = '1' and init_arready = '1' then
+ init_ar_hs <= init_ar_hs + 1;
+ end if;
+ if init_rvalid = '1' and init_rready = '1' then
+ init_r_hs <= init_r_hs + 1;
+ end if;
+ for i in config_array'range loop
+ if responders_write_address_valid(i) = '1' and responders_write_address_ready(i) = '1' then
+ resp_aw_hs <= resp_aw_hs + 1;
+ end if;
+ if responders_read_address_valid(i) = '1' and responders_read_address_ready(i) = '1' then
+ resp_ar_hs <= resp_ar_hs + 1;
+ end if;
+ end loop;
+ end if;
+ end process;
+
+ -- AXI payload stability: while a VALID is asserted without its READY, the
+ -- payload it carries must not change.
+ stability: process(clk)
+ variable prev_awaddr : std_logic_vector(init_awaddr'range);
+ variable prev_araddr : std_logic_vector(init_araddr'range);
+ variable prev_rdata : std_logic_vector(init_rdata'range);
+ variable had_aw : boolean := false;
+ variable had_ar : boolean := false;
+ variable had_r : boolean := false;
+ begin
+ if rising_edge(clk) then
+ if reset = '0' and man_mode = '0' then
+ if had_aw and init_awvalid = '1' then
+ check_equal(init_awaddr, prev_awaddr, "AWADDR changed while AWVALID was stalled");
+ end if;
+ if had_ar and init_arvalid = '1' then
+ check_equal(init_araddr, prev_araddr, "ARADDR changed while ARVALID was stalled");
+ end if;
+ if had_r and init_rvalid = '1' then
+ check_equal(init_rdata, prev_rdata, "RDATA changed while RVALID was stalled");
+ end if;
+ end if;
+
+ had_aw := init_awvalid = '1' and init_awready = '0';
+ prev_awaddr := init_awaddr;
+ had_ar := init_arvalid = '1' and init_arready = '0';
+ prev_araddr := init_araddr;
+ had_r := init_rvalid = '1' and init_rready = '0';
+ prev_rdata := init_rdata;
+ end if;
+ end process;
+
+end th;
diff --git a/hdl/ip/vhd/axi_blocks/sims/axil_slow_responder.vhd b/hdl/ip/vhd/axi_blocks/sims/axil_slow_responder.vhd
new file mode 100644
index 00000000..7542b9cb
--- /dev/null
+++ b/hdl/ip/vhd/axi_blocks/sims/axil_slow_responder.vhd
@@ -0,0 +1,211 @@
+-- This Source Code Form is subject to the terms of the Mozilla Public
+-- License, v. 2.0. If a copy of the MPL was not distributed with this
+-- file, You can obtain one at https://mozilla.org/MPL/2.0/.
+
+library ieee;
+use ieee.std_logic_1164.all;
+use ieee.numeric_std.all;
+
+use work.axil_common_pkg.all;
+
+-- Sim-only AXI-Lite responder that deliberately does *not* look like
+-- axil_target_txn. Every responder in the tree today ties wready to awready and
+-- makes bvalid a single-cycle pulse, which means the interconnect has only ever
+-- seen one handshake shape. This model:
+-- * accepts AW and W independently, each after its own stall
+-- * delays B and R by a variable number of cycles after the address phase
+-- * holds BVALID/RVALID until the corresponding READY
+-- The stalls come from an LFSR so the pattern varies across a test but is
+-- identical run to run.
+
+entity axil_slow_responder is
+ generic (
+ addr_width : integer := 8;
+ -- Longest stall this model will insert on any channel
+ max_delay : integer := 3;
+ -- LFSR seed, so multiple instances can stall differently
+ seed : std_logic_vector(7 downto 0) := x"A5"
+ );
+ port (
+ clk : in std_logic;
+ reset : in std_logic;
+
+ awvalid : in std_logic;
+ awready : out std_logic;
+ awaddr : in std_logic_vector(addr_width - 1 downto 0);
+
+ wvalid : in std_logic;
+ wready : out std_logic;
+ wdata : in std_logic_vector(31 downto 0);
+ wstrb : in std_logic_vector(3 downto 0);
+
+ bvalid : out std_logic;
+ bready : in std_logic;
+ bresp : out std_logic_vector(1 downto 0);
+
+ arvalid : in std_logic;
+ arready : out std_logic;
+ araddr : in std_logic_vector(addr_width - 1 downto 0);
+
+ rvalid : out std_logic;
+ rready : in std_logic;
+ rdata : out std_logic_vector(31 downto 0);
+ rresp : out std_logic_vector(1 downto 0)
+ );
+end entity;
+
+architecture rtl of axil_slow_responder is
+
+ constant NUM_WORDS : integer := 16;
+
+ type storage_t is array (0 to NUM_WORDS - 1) of std_logic_vector(31 downto 0);
+
+ signal storage : storage_t;
+ signal lfsr : std_logic_vector(7 downto 0);
+
+ signal aw_done : std_logic;
+ signal w_done : std_logic;
+ signal ar_done : std_logic;
+
+ signal aw_cnt : integer range 0 to max_delay;
+ signal w_cnt : integer range 0 to max_delay;
+ signal b_cnt : integer range 0 to max_delay;
+ signal ar_cnt : integer range 0 to max_delay;
+ signal r_cnt : integer range 0 to max_delay;
+
+ signal awaddr_reg : std_logic_vector(addr_width - 1 downto 0);
+ signal wdata_reg : std_logic_vector(31 downto 0);
+ signal wstrb_reg : std_logic_vector(3 downto 0);
+ signal rdata_reg : std_logic_vector(31 downto 0);
+
+ -- The LFSR fields are two bits, so a caller asking for a shorter longest
+ -- stall than 3 needs the value clamped to stay inside the counter ranges.
+ function delay_of (constant bits : std_logic_vector(1 downto 0)) return integer is
+ variable value : integer;
+ begin
+ value := to_integer(unsigned(bits));
+ if value > max_delay then
+ value := max_delay;
+ end if;
+ return value;
+ end function;
+
+begin
+
+ bresp <= OKAY;
+ rresp <= OKAY;
+ rdata <= rdata_reg;
+
+ lfsr_gen: process(clk, reset)
+ begin
+ if reset = '1' then
+ lfsr <= seed;
+ elsif rising_edge(clk) then
+ lfsr <= lfsr(6 downto 0) & (lfsr(7) xor lfsr(5) xor lfsr(4) xor lfsr(3));
+ end if;
+ end process;
+
+ txn: process(clk, reset)
+ variable idx : integer range 0 to NUM_WORDS - 1;
+ begin
+ if reset = '1' then
+ storage <= (others => (others => '0'));
+ awready <= '0';
+ wready <= '0';
+ bvalid <= '0';
+ arready <= '0';
+ rvalid <= '0';
+ aw_done <= '0';
+ w_done <= '0';
+ ar_done <= '0';
+ aw_cnt <= 0;
+ w_cnt <= delay_of("01");
+ b_cnt <= delay_of("01");
+ ar_cnt <= 0;
+ r_cnt <= delay_of("01");
+ elsif rising_edge(clk) then
+ -- readys are single cycle pulses
+ awready <= '0';
+ wready <= '0';
+ arready <= '0';
+
+ -- capture whatever handshaked this cycle
+ if awready = '1' and awvalid = '1' then
+ aw_done <= '1';
+ awaddr_reg <= awaddr;
+ end if;
+ if wready = '1' and wvalid = '1' then
+ w_done <= '1';
+ wdata_reg <= wdata;
+ wstrb_reg <= wstrb;
+ end if;
+ if arready = '1' and arvalid = '1' then
+ ar_done <= '1';
+ idx := to_integer(unsigned(araddr(5 downto 2)));
+ rdata_reg <= storage(idx);
+ end if;
+
+ -- stall, then accept. AW and W are completely independent.
+ if aw_done = '0' and awready = '0' and awvalid = '1' then
+ if aw_cnt = 0 then
+ awready <= '1';
+ else
+ aw_cnt <= aw_cnt - 1;
+ end if;
+ end if;
+ if w_done = '0' and wready = '0' and wvalid = '1' then
+ if w_cnt = 0 then
+ wready <= '1';
+ else
+ w_cnt <= w_cnt - 1;
+ end if;
+ end if;
+ if ar_done = '0' and arready = '0' and arvalid = '1' then
+ if ar_cnt = 0 then
+ arready <= '1';
+ else
+ ar_cnt <= ar_cnt - 1;
+ end if;
+ end if;
+
+ -- responses, held until ready
+ if aw_done = '1' and w_done = '1' and bvalid = '0' then
+ if b_cnt = 0 then
+ bvalid <= '1';
+ idx := to_integer(unsigned(awaddr_reg(5 downto 2)));
+ for byte in 0 to 3 loop
+ if wstrb_reg(byte) = '1' then
+ storage(idx)(8 * byte + 7 downto 8 * byte) <= wdata_reg(8 * byte + 7 downto 8 * byte);
+ end if;
+ end loop;
+ else
+ b_cnt <= b_cnt - 1;
+ end if;
+ end if;
+ if ar_done = '1' and rvalid = '0' then
+ if r_cnt = 0 then
+ rvalid <= '1';
+ else
+ r_cnt <= r_cnt - 1;
+ end if;
+ end if;
+
+ -- teardown, and pick the next set of stalls
+ if bvalid = '1' and bready = '1' then
+ bvalid <= '0';
+ aw_done <= '0';
+ w_done <= '0';
+ aw_cnt <= delay_of(lfsr(1 downto 0));
+ w_cnt <= delay_of(lfsr(3 downto 2));
+ b_cnt <= delay_of(lfsr(5 downto 4));
+ end if;
+ if rvalid = '1' and rready = '1' then
+ rvalid <= '0';
+ ar_done <= '0';
+ ar_cnt <= delay_of(lfsr(4 downto 3));
+ r_cnt <= delay_of(lfsr(7 downto 6));
+ end if;
+ end if;
+ end process;
+
+end rtl;
diff --git a/hdl/ip/vhd/axi_blocks/sims/axil_sram_responder.vhd b/hdl/ip/vhd/axi_blocks/sims/axil_sram_responder.vhd
new file mode 100644
index 00000000..2e2d1a82
--- /dev/null
+++ b/hdl/ip/vhd/axi_blocks/sims/axil_sram_responder.vhd
@@ -0,0 +1,113 @@
+-- This Source Code Form is subject to the terms of the Mozilla Public
+-- License, v. 2.0. If a copy of the MPL was not distributed with this
+-- file, You can obtain one at https://mozilla.org/MPL/2.0/.
+
+library ieee;
+use ieee.std_logic_1164.all;
+use ieee.numeric_std.all;
+
+use work.axil_common_pkg.all;
+
+-- Sim-only AXI-Lite responder: a handful of read/write words behind the *real*
+-- axil_target_txn block. Using the production transaction block is the whole
+-- point of this model -- it reproduces the handshake contract that every
+-- register block in the tree presents to the interconnect:
+-- * awready is a registered one-cycle pulse gated on awvalid AND wvalid
+-- * wready is tied to awready, so AW and W always handshake together
+-- * arready is combinational (not rvalid), so AR handshakes immediately
+-- * bvalid is a single-cycle pulse when bready is already asserted
+-- That last property in particular catches a pipeline stage that goes looking
+-- for B a cycle too late.
+
+entity axil_sram_responder is
+ generic (
+ addr_width : integer := 8
+ );
+ port (
+ clk : in std_logic;
+ reset : in std_logic;
+
+ awvalid : in std_logic;
+ awready : out std_logic;
+ awaddr : in std_logic_vector(addr_width - 1 downto 0);
+
+ wvalid : in std_logic;
+ wready : out std_logic;
+ wdata : in std_logic_vector(31 downto 0);
+ wstrb : in std_logic_vector(3 downto 0);
+
+ bvalid : out std_logic;
+ bready : in std_logic;
+ bresp : out std_logic_vector(1 downto 0);
+
+ arvalid : in std_logic;
+ arready : out std_logic;
+ araddr : in std_logic_vector(addr_width - 1 downto 0);
+
+ rvalid : out std_logic;
+ rready : in std_logic;
+ rdata : out std_logic_vector(31 downto 0);
+ rresp : out std_logic_vector(1 downto 0)
+ );
+end entity;
+
+architecture rtl of axil_sram_responder is
+
+ constant NUM_WORDS : integer := 16;
+
+ type storage_t is array (0 to NUM_WORDS - 1) of std_logic_vector(31 downto 0);
+
+ signal storage : storage_t;
+ signal active_read : std_logic;
+ signal active_write : std_logic;
+ signal rdata_reg : std_logic_vector(31 downto 0);
+
+begin
+
+ axil_target_txn_inst: entity work.axil_target_txn
+ port map (
+ clk => clk,
+ reset => reset,
+ arvalid => arvalid,
+ arready => arready,
+ awvalid => awvalid,
+ awready => awready,
+ wready => wready,
+ wvalid => wvalid,
+ bvalid => bvalid,
+ bready => bready,
+ bresp => bresp,
+ rvalid => rvalid,
+ rready => rready,
+ rresp => rresp,
+ active_read => active_read,
+ active_write => active_write
+ );
+
+ rdata <= rdata_reg;
+
+ -- word addressed off bits 5:2, so 16 words repeat through the space
+ sram: process(clk, reset)
+ variable idx : integer range 0 to NUM_WORDS - 1;
+ begin
+ if reset = '1' then
+ storage <= (others => (others => '0'));
+ rdata_reg <= (others => '0');
+ elsif rising_edge(clk) then
+ if active_write = '1' then
+ idx := to_integer(unsigned(awaddr(5 downto 2)));
+ for byte in 0 to 3 loop
+ if wstrb(byte) = '1' then
+ storage(idx)(8 * byte + 7 downto 8 * byte) <= wdata(8 * byte + 7 downto 8 * byte);
+ end if;
+ end loop;
+ end if;
+
+ if active_read = '1' then
+ idx := to_integer(unsigned(araddr(5 downto 2)));
+ rdata_reg <= storage(idx);
+ end if;
+ end if;
+ end process;
+
+end rtl;
diff --git a/hdl/ip/vhd/spi_nor_controller/sims/spi_nor_th.vhd b/hdl/ip/vhd/spi_nor_controller/sims/spi_nor_th.vhd
index 396574bb..4e275c1f 100644
--- a/hdl/ip/vhd/spi_nor_controller/sims/spi_nor_th.vhd
+++ b/hdl/ip/vhd/spi_nor_controller/sims/spi_nor_th.vhd
@@ -32,7 +32,7 @@ architecture th of spi_nor_th is
signal io_oe : std_logic_vector(3 downto 0);
constant config_array : axil_responder_cfg_array_t(0 downto 0) :=
(
- 0 => (base_addr => x"00000100", addr_span_bits => 8)
+ 0 => resp_cfg(base_addr => x"00000100", addr_span_bits => 8)
);
signal responders : axil32x32_pkg.axil_array_t(0 downto 0);
signal responders_8b : axil8x32_pkg.axil_array_t(0 downto 0);
diff --git a/hdl/projects/cosmo_hp/cosmo_hp_top.vhd b/hdl/projects/cosmo_hp/cosmo_hp_top.vhd
index 8277ba1e..0a91d227 100644
--- a/hdl/projects/cosmo_hp/cosmo_hp_top.vhd
+++ b/hdl/projects/cosmo_hp/cosmo_hp_top.vhd
@@ -205,10 +205,10 @@ architecture rtl of cosmo_hp_top is
signal pca_int_n : std_logic_vector(io_i2c_addr'length - 1 downto 0);
constant config_array : axil_responder_cfg_array_t :=
- (0 => (base_addr => x"00000000", addr_span_bits => 8),
- 1 => (base_addr => x"00000100", addr_span_bits => 8),
- 2 => (base_addr => x"00000200", addr_span_bits => 8),
- 3 => (base_addr => x"00000300", addr_span_bits => 8)
+ (0 => resp_cfg(base_addr => x"00000000", addr_span_bits => 8),
+ 1 => resp_cfg(base_addr => x"00000100", addr_span_bits => 8),
+ 2 => resp_cfg(base_addr => x"00000200", addr_span_bits => 8),
+ 3 => resp_cfg(base_addr => x"00000300", addr_span_bits => 8)
);
signal sp_write_address_addr : std_logic_vector(15 downto 0);
diff --git a/hdl/projects/cosmo_seq/cosmo_seq_top.vhd b/hdl/projects/cosmo_seq/cosmo_seq_top.vhd
index 0b5eeb46..45893855 100644
--- a/hdl/projects/cosmo_seq/cosmo_seq_top.vhd
+++ b/hdl/projects/cosmo_seq/cosmo_seq_top.vhd
@@ -345,14 +345,17 @@ architecture rtl of cosmo_seq_top is
constant ESPI_RESP_IDX: integer := 7;
constant config_array : axil_responder_cfg_array_t :=
- (INFO_RESP_IDX => (base_addr => x"00000000", addr_span_bits => 8),
- SPINOR_RESP_IDX => (base_addr => x"00000100", addr_span_bits => 8),
- SEQ_RESP_IDX => (base_addr => x"00000200", addr_span_bits => 8),
- SP_I2C_RESP_IDX => (base_addr => x"00000300", addr_span_bits => 8),
- SP5_HP_RESP_IDX => (base_addr => x"00000400", addr_span_bits => 8),
- SPD_PROXY_RESP_IDX => (base_addr => x"00000500", addr_span_bits => 8),
- DBG_CTRL_RESP_IDX => (base_addr => x"00000600", addr_span_bits => 8),
- ESPI_RESP_IDX => (base_addr => x"00008000", addr_span_bits => 15)
+ (INFO_RESP_IDX => resp_cfg(base_addr => x"00000000", addr_span_bits => 8),
+ SPINOR_RESP_IDX => resp_cfg(base_addr => x"00000100", addr_span_bits => 8),
+ SEQ_RESP_IDX => resp_cfg(base_addr => x"00000200", addr_span_bits => 8),
+ SP_I2C_RESP_IDX => resp_cfg(base_addr => x"00000300", addr_span_bits => 8),
+ SP5_HP_RESP_IDX => resp_cfg(base_addr => x"00000400", addr_span_bits => 8),
+ SPD_PROXY_RESP_IDX => resp_cfg(base_addr => x"00000500", addr_span_bits => 8),
+ DBG_CTRL_RESP_IDX => resp_cfg(base_addr => x"00000600", addr_span_bits => 8),
+ -- eSPI is the largest register file and the most distant block, and it
+ -- owns the worst 125MHz path in the design, so give the fabric a cycle
+ -- in each direction to get there and back.
+ ESPI_RESP_IDX => resp_cfg(base_addr => x"00008000", addr_span_bits => 15, pipe_stages => 1)
);
signal fmc_axi_if : axil26x32_pkg.axil_t;
signal fabric_responders : axil32x32_pkg.axil_array_t(config_array'range);
diff --git a/hdl/projects/grapefruit/grapefruit_top.vhd b/hdl/projects/grapefruit/grapefruit_top.vhd
index ec1b8add..5d70aad9 100644
--- a/hdl/projects/grapefruit/grapefruit_top.vhd
+++ b/hdl/projects/grapefruit/grapefruit_top.vhd
@@ -205,10 +205,10 @@ architecture rtl of grapefruit_top is
-- TODO: someday I'd like the rdl stuff to both generate this and the fabric maybe?
constant config_array : axil_responder_cfg_array_t :=
- (0 => (base_addr => x"00000000", addr_span_bits => 8),
- 1 => (base_addr => x"00000100", addr_span_bits => 8),
- 2 => (base_addr => x"00000200", addr_span_bits => 8),
- 3 => (base_addr => x"00008000", addr_span_bits => 15)
+ (0 => resp_cfg(base_addr => x"00000000", addr_span_bits => 8),
+ 1 => resp_cfg(base_addr => x"00000100", addr_span_bits => 8),
+ 2 => resp_cfg(base_addr => x"00000200", addr_span_bits => 8),
+ 3 => resp_cfg(base_addr => x"00008000", addr_span_bits => 15)
);
signal fabric_responders : axil32x32_pkg.axil_array_t(config_array'range);
signal responders_8b : axil8x32_pkg.axil_array_t(config_array'range);