153 """Decoupled, pipelinable grant scheduler (`pipelined_scheduler=True`).
155 A **grant queue** holds upcoming winners for the datapath to pop, and a
156 **sweep scheduler** refills it off the critical path. That breaks the flat
157 arbiter's single-cycle `grant -> grant` loop, its dominant timing limiter at
158 high fan-in. A queued entry is a hint about who to serve next, not a promise
159 that a particular message is waiting: an entry whose input has since gone
160 idle is skipped in one cycle (`stale` below) rather than stalling the output.
162 Consequently service order is best-effort, and `queue_depth` bounds only how
163 far ahead of the datapath decisions are committed -- it is not a fairness
164 knob. See section 7.1 of `docs/components/ChannelArbiter.md` for why
165 committing early is safe and for the full ordering/latency caveats."""
166 assert num_inputs >= 2,
"GrantSchedulerMod requires at least two inputs"
167 gw = clog2(num_inputs)
169 class GrantScheduler(Module):
174 valids = Input(Bits(num_inputs))
176 launch = Input(Bits(1))
178 msg_end = Input(Bits(1))
181 grant = Output(Bits(gw))
183 grant_oh = Output(Bits(num_inputs))
185 busy = Output(Bits(1))
187 switch = Output(Bits(1))
189 next_grant_oh = Output(Bits(num_inputs))
190 next_busy = Output(Bits(1))
193 def build(ports) -> None:
196 next_grant = Wire(Bits(gw),
"next_grant")
197 next_busy = Wire(Bits(1),
"next_busy")
199 next_grant, next_busy)
203 gq = SeqFIFO(Bits(gw), queue_depth, clk, rst)
204 q_pop = Wire(Bits(1),
"gq_pop")
205 q_head = gq.pop(q_pop)
206 q_nonempty = ~gq.empty
209 pending = Reg(Bits(num_inputs),
213 name=
"sched_pending")
214 pend_nonzero = pending != Bits(num_inputs)(0)
216 neg_pending = ((~pending).as_uint(num_inputs) +
217 UInt(num_inputs)(1)).as_bits(num_inputs)
218 low = pending & neg_pending
219 push = pend_nonzero & ~gq.full
230 cleared = pending & ~low
231 sweep_done = push & (cleared == Bits(num_inputs)(0))
233 Mux(~pend_nonzero | sweep_done, Mux(push, pending, cleared),
239 started = Reg(Bits(1), clk, rst, rst_value=0, name=
"grant_started")
240 sel_valid_now = (ports.valids & grant_oh).or_reduce()
243 stale = busy & ~started & ~sel_valid_now & q_nonempty
244 advance = ports.msg_end | stale
245 take_next = ~busy | advance
246 q_pop.assign(take_next & q_nonempty)
248 next_grant.assign(Mux(take_next & q_nonempty, grant, q_head))
249 next_busy.assign(Mux(take_next, busy, q_nonempty))
252 started_next = Mux(ports.launch, started, Bits(1)(1))
253 started.assign(Mux(take_next, started_next, Bits(1)(0)))
258 return GrantScheduler
325 """Flat round-robin grant control (the default strategy).
327 Answers "who is granted next?" combinationally in the cycle the current
328 message ends, using two `RoundRobinArbiter` instances -- one for picking up
329 from idle, one for the message-end turnaround -- plus the `rr_ptr` fairness
330 pointer, which is private to this strategy. See section 7 of
331 `docs/components/ChannelArbiter.md`.
333 `launch` is unused; it exists only to match `GrantSchedulerMod`'s
335 assert num_inputs >= 2,
"RoundRobinControlMod requires at least two inputs"
336 gw = clog2(num_inputs)
338 class RoundRobinControl(Module):
342 valids = Input(Bits(num_inputs))
343 launch = Input(Bits(1))
344 msg_end = Input(Bits(1))
346 grant = Output(Bits(gw))
347 grant_oh = Output(Bits(num_inputs))
348 busy = Output(Bits(1))
349 switch = Output(Bits(1))
351 next_grant_oh = Output(Bits(num_inputs))
352 next_busy = Output(Bits(1))
355 def build(ports) -> None:
358 next_grant = Wire(Bits(gw),
"next_grant")
359 next_busy = Wire(Bits(1),
"next_busy")
361 next_grant, next_busy)
362 rr_ptr = Reg(Bits(gw), clk, rst, name=
"rr_ptr")
365 def round_robin(valids_vec: BitsSignal, start: BitsSignal,
366 name: str) -> Tuple[BitsSignal, BitsSignal]:
367 """Instantiate a RoundRobinArbiter over `valids_vec` starting from
369 inst = rr_arbiter(valids=valids_vec, start=start, instance_name=name)
370 return inst.winner, inst.any_valid
372 grant_u = grant.as_uint(gw)
373 is_last_idx = grant == Bits(gw)(num_inputs - 1)
374 grant_p1 = Mux(is_last_idx, (grant_u + UInt(gw)(1)).as_bits(gw),
377 winner_idle, any_idle = round_robin(ports.valids, rr_ptr,
"rr_idle")
384 valids_next = ports.valids & ~grant_oh
385 winner_next, any_next = round_robin(valids_next, grant_p1,
"rr_next")
387 pick = ~busy & any_idle
388 reend = busy & ports.msg_end
389 grant_if_not_reend = Mux(pick, grant, winner_idle)
390 busy_if_not_reend = Mux(pick, busy, Bits(1)(1))
392 next_grant.assign(Mux(reend, grant_if_not_reend, winner_next))
393 next_busy.assign(Mux(reend, busy_if_not_reend, any_next))
394 rr_ptr.assign(Mux(reend, rr_ptr, grant_p1))
395 ports.switch = pick | (reend & any_next)
397 return RoundRobinControl
401def ChannelArbiterMod(channel_type: Channel,
403 output_fifo_depth: int,
406 mux_pipeline_levels: Optional[int],
407 pipelined_scheduler: bool,
408 grant_queue_depth: int,
409 wide_fanin: Optional[bool] =
None):
410 """Build a pipelined, list-aware N:1 channel multiplexer module. See the
411 `ChannelArbiter` convenience function for the user-facing entry point and
412 `docs/components/ChannelArbiter.md` for the design."""
414 assert num_inputs >= 2,
"ChannelArbiterMod requires at least two inputs"
415 if wide_fanin
is None:
416 wide_fanin = num_inputs > _WIDE_FANIN_THRESHOLD
417 inner = channel_type.inner_type
421 is_window = isinstance(inner, Window)
423 lowered = inner.lowered_type
424 field_names = [n
for n, _
in lowered.fields]
if isinstance(
425 lowered, StructType)
else None
426 if field_names
is None or "last" not in field_names:
428 "ChannelArbiter can only auto-detect list framing for window types "
429 "whose lowered frame is a struct with a 'last' field; got lowered "
430 f
"type {lowered}. (Serial/union-framed windows are not supported.)")
431 width = lowered.bitwidth
433 width = inner.bitwidth
436 f
"ChannelArbiter requires a fixed-width payload; got {inner}")
442 beat_type = Bits(width)
446 gw = clog2(num_inputs)
450 mux_pipeline_levels))
453 pipe_latency = tree_latency + 1
454 if output_fifo_depth
is not None and \
455 output_fifo_depth <= pipe_latency:
457 f
"output_fifo_depth ({output_fifo_depth}) must be > the pipeline "
458 f
"latency ({pipe_latency})")
460 class ChannelArbiterImpl(Module):
468 inputs = Input(Array(channel_type, num_inputs))
469 output = Output(channel_type)
472 def build(ports) -> None:
476 depth = (pipe_latency + ChannelArbiterImpl._SLACK
477 if output_fifo_depth
is None else output_fifo_depth)
478 cw = max(1, depth.bit_length())
482 def flit_last(typed_sig: Signal) -> BitsSignal:
483 """High when 'typed_sig' is the last flit of its message."""
485 return typed_sig.unwrap()[
"last"]
488 def to_bits(typed_sig: Signal) -> BitsSignal:
489 """Bitcast the payload to raw bits for the datapath."""
491 typed_sig = typed_sig.unwrap()
492 return typed_sig.bitcast(Bits(width))
494 def from_bits(bits: BitsSignal) -> Signal:
495 """Reconstruct the payload from raw bits for the output channel."""
497 return inner.wrap(bits.bitcast(inner.lowered_type))
498 return bits.bitcast(inner)
503 grant = Wire(Bits(gw),
"grant")
504 grant_oh = Wire(Bits(num_inputs),
"grant_oh")
505 busy = Wire(Bits(1),
"busy")
507 next_grant_oh = Wire(Bits(num_inputs),
"next_grant_oh")
508 next_busy = Wire(Bits(1),
"next_busy_dp")
509 next_credit_gt0 = Wire(Bits(1),
"next_credit_gt0")
510 credit = Reg(UInt(cw), clk, rst, rst_value=depth, name=
"credit")
512 credit_gt0 = credit > UInt(cw)(0)
514 credit_gt1 = credit > UInt(cw)(1)
517 valids: List[BitsSignal] = []
518 last_bits: List[BitsSignal] = []
519 data_bits: List[BitsSignal] = []
520 for i
in range(num_inputs):
521 chan = ports.inputs[i]
523 chan = chan.buffer(clk, rst, stages=1)
530 ready_i = (next_busy & next_grant_oh[i] & next_credit_gt0).reg(
531 clk, rst, rst_value=0, name=f
"ready_{i}")
533 ready_i = busy & grant_oh[i] & credit_gt0
534 data_i, valid_i = chan.unwrap(ready_i)
535 valids.append(valid_i)
536 last_bits.append(flit_last(data_i))
537 data_bits.append(to_bits(data_i))
545 go = busy & credit_gt0
546 sel_valid = Or(*[valids[i] & grant_oh[i]
for i
in range(num_inputs)])
549 [valids[i] & last_bits[i] & grant_oh[i]
for i
in range(num_inputs)])
551 sel_valid = Mux(grant, *valids)
552 sel_last = Mux(grant, *last_bits)
554 sel_bits = Bits(0)(0)
556 sel_bits =
_select_mux(grant, data_bits, clk, rst, mux_pipeline_levels)
561 launch = go & sel_valid
562 msg_end = go & sel_valid_last
564 launch = busy & sel_valid & credit_gt0
565 msg_end = launch & sel_last
573 out_valid = credit < UInt(cw)(depth)
574 payload_bits = Bits(0)(0)
580 for _
in range(tree_latency):
581 pipe_valid = pipe_valid.reg(clk, rst)
582 pipe_valid = pipe_valid.reg(clk, rst, name=
"pipe_valid")
584 pipe_beat = sel_bits.reg(clk, name=
"pipe_beat")
585 fifo = SeqFIFO(beat_type, depth, clk, rst)
586 fifo.push(pipe_beat, pipe_valid)
587 fifo_pop = Wire(Bits(1),
"arb_pop")
588 out_valid = ~fifo.empty
589 payload_bits = fifo.pop(fifo_pop)
591 out_chan, out_ready = channel_type.wrap(from_bits(payload_bits),
593 ports.output = out_chan
594 pop = out_valid & out_ready
595 if fifo_pop
is not None:
599 next_credit = ((credit + pop.as_uint(cw)).as_uint(cw) -
600 launch.as_uint(cw)).as_uint(cw)
601 credit.assign(next_credit)
605 next_credit_gt0.assign(
606 Mux(pop ^ launch, credit_gt0, Mux(pop, credit_gt1,
614 ctrl = ctrl_mod(clk=clk,
616 valids=BitsSignal.concat(list(reversed(valids))),
619 instance_name=
"arb_ctrl")
620 grant.assign(ctrl.grant)
621 grant_oh.assign(ctrl.grant_oh)
622 busy.assign(ctrl.busy)
624 next_grant_oh.assign(ctrl.next_grant_oh)
625 next_busy.assign(ctrl.next_busy)
626 arb_switch = ctrl.switch
630 Telemetry.report_signal(clk, rst, AppID(
"selectedChannel"), grant)
631 Telemetry.report_signal(clk, rst, AppID(
"busy"), busy)
633 for i
in range(num_inputs):
634 served = Counter(64)(clk=clk,
637 increment=launch & grant_oh[i])
638 Telemetry.report_signal(clk, rst, AppID(f
"grantCount_{i}"),
641 total_flits = Counter(64)(clk=clk,
645 Telemetry.report_signal(clk, rst, AppID(
"totalFlits"), total_flits.out)
646 total_msgs = Counter(64)(clk=clk,
650 Telemetry.report_signal(clk, rst, AppID(
"totalMessages"),
652 arb_switches = Counter(64)(clk=clk,
655 increment=arb_switch)
656 Telemetry.report_signal(clk, rst, AppID(
"arbSwitches"),
660 cur_len = Counter(32)(clk=clk, rst=rst, clear=msg_end, increment=launch)
661 msg_len = (cur_len.out + UInt(32)(1)).as_uint(32)
662 max_len = Reg(UInt(32), clk, rst, rst_value=0, name=
"max_list_len")
663 is_new_max = msg_end & (msg_len > max_len)
664 max_len.assign(Mux(is_new_max, max_len, msg_len))
665 Telemetry.report_signal(clk, rst, AppID(
"maxListLen"), max_len)
668 occ = (UInt(cw)(depth) - credit).as_uint(cw)
669 inflight_hw = Reg(UInt(cw), clk, rst, rst_value=0, name=
"inflight_hw")
670 is_new_hw = occ > inflight_hw
671 inflight_hw.assign(Mux(is_new_hw, inflight_hw, occ))
672 Telemetry.report_signal(clk, rst, AppID(
"inflightHighWater"),
675 return ChannelArbiterImpl
678def ChannelArbiter(input_channels: List[ChannelSignal],
682 appid: Optional[AppID] =
None,
683 output_fifo_depth: Optional[int] =
None,
684 buffer_inputs: bool =
True,
685 mux_pipeline_levels: Optional[int] =
None,
686 pipelined_scheduler: bool =
False,
687 grant_queue_depth: int = 4,
688 wide_fanin: Optional[bool] =
None,
689 telemetry: bool =
True) -> ChannelSignal:
690 """Build a pipelined, list-aware N:1 channel multiplexer.
692 Unlike the combinational `pycde.esi.ChannelMux`, this is a flat registered
693 round-robin arbiter with a feed-forward output stage (output register + FIFO
694 + credit counter), so it closes timing at high fan-in. It also keeps
695 multi-flit list messages contiguous: once an input is granted, it holds the
696 output until a flit whose 'last' field is set has been transferred. List
697 framing is auto-detected from the channel type (window payloads with a 'last'
698 field); all other payloads are treated as single-flit messages.
701 input_channels: the channels to multiplex. All must share the same
703 clk, rst: clock and reset.
704 appid: optional `AppID` for the arbiter instance (e.g. to address it or to
705 disambiguate its telemetry in the appid hierarchy).
706 output_fifo_depth: depth of the output FIFO; must be greater than the
707 pipeline latency (one output register plus any selection-mux pipeline
708 latency). Defaults to that plus a small internal slack.
709 buffer_inputs: insert a per-input skid buffer to localize backpressure.
710 mux_pipeline_levels: if set, build the N:1 data-selection mux as an explicit
711 binary tree and insert a pipeline register after every this-many tree
712 levels (1 = register every level). This retimes the wide selection mux
713 for very large fan-in; the added latency is absorbed by the output FIFO /
714 credit counter. `None` (default) uses a flat combinational mux.
715 pipelined_scheduler: decouple grant selection from the datapath using a
716 grant queue fed by a sweep scheduler, instead of re-arbitrating
717 combinationally at each message end. This takes the round-robin tree out
718 of the single-cycle `grant -> grant` loop, which is the Fmax limiter at
719 high fan-in. Changes the service order (see `GrantSchedulerMod`).
720 grant_queue_depth: depth of that grant queue -- how many grant decisions
721 may be committed ahead of the datapath. Must be >= 2: a single entry
722 cannot keep the datapath fed back to back, so every message would cost a
723 refill bubble. This is not a fairness knob; a newly-valid input's wait
724 also scales with the number of concurrently active inputs (see
725 `GrantSchedulerMod`).
726 wide_fanin: timing structures for large fan-in (registered per-input
727 `ready`, one-hot loop selections); behaviour is unchanged. `None`
728 (default) enables them above `_WIDE_FANIN_THRESHOLD` inputs.
729 telemetry: emit telemetry (selected channel, list-length stats, etc.).
731 See `docs/components/ChannelArbiter.md`."""
733 assert len(input_channels) > 0
734 num_inputs = len(input_channels)
736 return input_channels[0]
738 channel_type = input_channels[0].type
739 for c
in input_channels:
740 if c.type != channel_type:
741 raise TypeError(
"All ChannelArbiter inputs must have the same type; got "
742 f
"{channel_type} and {c.type}")
743 if channel_type.signaling != ChannelSignaling.ValidReady:
744 raise TypeError(
"ChannelArbiter requires ValidReady channels; got "
747 if mux_pipeline_levels
is not None and mux_pipeline_levels < 1:
749 f
"mux_pipeline_levels must be >= 1, got {mux_pipeline_levels}")
757 if pipelined_scheduler
and grant_queue_depth < 2:
758 raise ValueError(f
"grant_queue_depth must be >= 2, got {grant_queue_depth}")
760 if wide_fanin
is None:
761 wide_fanin = num_inputs > _WIDE_FANIN_THRESHOLD
762 mod = ChannelArbiterMod(channel_type, num_inputs, output_fifo_depth,
763 buffer_inputs, telemetry, mux_pipeline_levels,
764 pipelined_scheduler, grant_queue_depth, wide_fanin)
765 inputs_array = Array(channel_type, num_inputs)(input_channels)
766 inst = mod(clk=clk, rst=rst, inputs=inputs_array, appid=appid)