Snitch Runtime
Loading...
Searching...
No Matches
dma.h File Reference

This file provides functions to program the Snitch DMA. More...

#include <string.h>
#include <math.h>
#include "idma_compute.h"

Go to the source code of this file.

Functions

uint32_t snrt_dma_start_1d (uint64_t dst, uint64_t src, size_t size, uint32_t channel)
 Start an asynchronous 1D DMA transfer with 64-bit wide pointers on a specific DMA channel.
 
uint32_t snrt_dma_start_1d (volatile void *dst, volatile void *src, size_t size, uint32_t channel=0)
 Start an asynchronous 1D DMA transfer using native-size pointers.
 
void snrt_dma_set_awuser (uint64_t field)
 Set AW user field of the DMA's AXI interface.
 
void snrt_dma_enable_multicast (uint64_t mask)
 Enable multicast for successive transfers.
 
void snrt_dma_enable_reduction (uint64_t mask, snrt_collective_opcode_t opcode)
 Enable reduction operations for successive transfers.
 
void snrt_dma_disable_multicast ()
 Disable multicast for successive transfers.
 
void snrt_dma_disable_reduction ()
 Disable reduction operations for successive transfers.
 
uint32_t snrt_dma_start_1d_reduction (uint64_t dst, uint64_t src, size_t size, uint64_t mask, snrt_collective_opcode_t opcode, uint32_t channel=0)
 Start an asynchronous reduction 1D DMA transfer with 64-bit wide pointers.
 
uint32_t snrt_dma_start_1d_reduction (uint64_t dst, uint64_t src, size_t size, snrt_comm_t comm, snrt_collective_opcode_t opcode, uint32_t channel=0)
 Start an asynchronous reduction 1D DMA transfer with 64-bit wide pointers.
 
uint32_t snrt_dma_start_1d_mcast (uint64_t dst, uint64_t src, size_t size, uint64_t mask, uint32_t channel=0)
 Start an asynchronous multicast 1D DMA transfer with 64-bit wide pointers.
 
uint32_t snrt_dma_start_1d_mcast (uint64_t dst, uint64_t src, size_t size, snrt_comm_t comm, uint32_t channel=0)
 Start an asynchronous multicast 1D DMA transfer with 64-bit wide pointers.
 
uint32_t snrt_dma_start_1d_reduction (volatile void *dst, volatile void *src, size_t size, uint64_t mask, snrt_collective_opcode_t opcode, uint32_t channel=0)
 Start an asynchronous reduction 1D DMA transfer using native-size pointers.
 
uint32_t snrt_dma_start_1d_mcast (volatile void *dst, volatile void *src, size_t size, uint64_t mask, uint32_t channel=0)
 Start an asynchronous multicast 1D DMA transfer using native-size pointers.
 
snrt_dma_txid_t snrt_dma_start_2d (uint64_t dst, uint64_t src, size_t size, size_t dst_stride, size_t src_stride, size_t repeat, uint32_t channel)
 Start an asynchronous 2D DMA transfer with 64-bit wide pointers.
 
uint32_t snrt_dma_start_2d (volatile void *dst, volatile void *src, size_t size, size_t dst_stride, size_t src_stride, size_t repeat, uint32_t channel=0)
 Start an asynchronous 2D DMA transfer using native-size pointers.
 
uint32_t snrt_dma_start_2d_mcast (uint64_t dst, uint64_t src, size_t size, size_t dst_stride, size_t src_stride, size_t repeat, uint32_t mask, uint32_t channel=0)
 Start an asynchronous, multicast 2D DMA transfer with 64-bit wide pointers.
 
uint32_t snrt_dma_start_2d_mcast (volatile void *dst, volatile void *src, size_t size, size_t dst_stride, size_t src_stride, size_t repeat, uint32_t mask, uint32_t channel=0)
 Start an asynchronous, multicast 2D DMA transfer using native-size pointers.
 
static uint32_t snrt_dma_busy (const uint32_t channel)
 Read DMA busy flag.
 
static uint32_t snrt_dma_would_block (const uint32_t channel)
 Read DMA would_block flag.
 
static void snrt_dma_wait (snrt_dma_txid_t txid, const uint32_t channel=0)
 Block until a DMA transfer finishes on a specific DMA channel.
 
static void snrt_dma_wait_all (const uint32_t channel=0)
 Block until a specific DMA channel is idle.
 
void snrt_dma_wait_all_channels (uint32_t num_channels)
 Block until the first num_channels channels are idle.
 
void snrt_dma_start_tracking ()
 Start tracking of dma performance region. Does not have any implications on the HW. Only injects a marker in the DMA traces that can be analyzed.
 
void snrt_dma_stop_tracking ()
 Stop tracking of dma performance region. Does not have any implications on the HW. Only injects a marker in the DMA traces that can be analyzed.
 
void snrt_dma_memset (void *ptr, uint8_t value, uint32_t len)
 Fast memset function performed by DMA.
 
snrt_dma_txid_t snrt_dma_load_1d_tile (volatile void *dst, volatile void *src, size_t tile_idx, size_t tile_size, uint32_t prec)
 Load a tile of a 1D array.
 
snrt_dma_txid_t snrt_dma_load_1d_tile_mcast (void *dst, void *src, size_t tile_idx, size_t tile_size, uint32_t prec, uint64_t mask)
 Load a tile of a 1D array.
 
snrt_dma_txid_t snrt_dma_reduction_load_1d_tile (void *dst, void *src, size_t tile_idx, size_t tile_size, uint32_t prec, uint64_t mask, snrt_collective_opcode_t opcode)
 Load a tile of a 1D array.
 
snrt_dma_txid_t snrt_dma_1d_to_2d (volatile void *dst, volatile void *src, size_t size, size_t row_size, size_t stride)
 Transfer and reshape a 1D array into a 2D array.
 
snrt_dma_txid_t snrt_dma_2d_to_1d (volatile void *dst, volatile void *src, size_t size, size_t row_size, size_t stride)
 Transfer and reshape a 2D array into a 1D array.
 
snrt_dma_txid_t snrt_dma_store_1d_tile (void *dst, void *src, size_t tile_idx, size_t tile_size, uint32_t prec)
 Store a tile to a 1D array.
 
snrt_dma_txid_t snrt_dma_load_2d_tile (void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, size_t tile_ld)
 Load a 2D tile of a 2D array.
 
snrt_dma_txid_t snrt_dma_load_2d_tile (void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec)
 Load a 2D tile of a 2D array.
 
snrt_dma_txid_t snrt_dma_load_2d_tile_mcast (void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, size_t tile_ld, uint32_t mask)
 Load a 2D tile of a 2D array using multicast.
 
snrt_dma_txid_t snrt_dma_load_2d_tile_mcast (void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, uint32_t mask)
 Load a 2D tile of a 2D array.
 
snrt_dma_txid_t snrt_dma_load_2d_tile_mcast (void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, snrt_comm_t comm)
 Load a 2D tile of a 2D array using multicast.
 
snrt_dma_txid_t snrt_dma_load_2d_tile_in_banks (void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, size_t num_banks)
 Load a 2D tile of a 2D array and reshape it to occupy a subset of TCDM banks.
 
snrt_dma_txid_t snrt_dma_store_2d_tile (void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, size_t tile_ld)
 Store a 2D tile to a 2D array.
 
snrt_dma_txid_t snrt_dma_store_2d_tile (void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec)
 Store a 2D tile of a 2D array.
 
snrt_dma_txid_t snrt_dma_store_2d_tile_from_banks (void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, size_t num_banks)
 Store a 2D tile of a 2D array from a 1D layout occupying a subset of TCDM banks.
 
void snrt_dma_set_opcode_params (uint32_t opcode, uint32_t params)
 Set the on-the-fly compute configuration for subsequent transfers.
 
void snrt_dma_set_opcode (uint32_t opcode)
 Set a parameterless on-the-fly compute op for subsequent transfers.
 
void snrt_dma_enable_transpose (uint32_t mode, uint32_t tensor_m, uint32_t tensor_n)
 Enable the tiled transpose of a row-major tensor for successive transfers.
 
void snrt_dma_disable_compute ()
 Disable on-the-fly compute for successive transfers.
 
uint32_t snrt_dma_start_transpose (uint64_t dst, uint64_t src, size_t size, uint32_t mode, uint32_t tensor_m, uint32_t tensor_n, uint32_t channel=0)
 Start an asynchronous transposing DMA transfer.
 
uint32_t snrt_dma_start_transpose (volatile void *dst, volatile void *src, size_t size, uint32_t mode, uint32_t tensor_m, uint32_t tensor_n, uint32_t channel=0)
 Start an asynchronous transposing DMA transfer.
 

Detailed Description

This file provides functions to program the Snitch DMA.

Function Documentation

◆ snrt_dma_1d_to_2d()

snrt_dma_txid_t snrt_dma_1d_to_2d ( volatile void * dst,
volatile void * src,
size_t size,
size_t row_size,
size_t stride )
inline

Transfer and reshape a 1D array into a 2D array.

Parameters
dstPointer to the destination array.
srcPointer to the source array.
sizeNumber of bytes to transfer.
row_sizeSize of a row in the 2D array, in bytes.
strideStride between successive rows in the 2D array, in bytes.
531 {
532 return snrt_dma_start_2d(dst, src, row_size, stride, row_size,
533 size / row_size);
534}
snrt_dma_txid_t snrt_dma_start_2d(uint64_t dst, uint64_t src, size_t size, size_t dst_stride, size_t src_stride, size_t repeat, uint32_t channel)
Start an asynchronous 2D DMA transfer with 64-bit wide pointers.
Definition dma.h:235

◆ snrt_dma_2d_to_1d()

snrt_dma_txid_t snrt_dma_2d_to_1d ( volatile void * dst,
volatile void * src,
size_t size,
size_t row_size,
size_t stride )
inline

Transfer and reshape a 2D array into a 1D array.

Parameters
dstPointer to the destination array.
srcPointer to the source array.
sizeNumber of bytes to transfer.
row_sizeSize of a row in the 2D array, in bytes.
strideStride between successive rows in the 2D array, in bytes.
546 {
547 return snrt_dma_start_2d(dst, src, row_size, row_size, stride,
548 size / row_size);
549}

◆ snrt_dma_busy()

static uint32_t snrt_dma_busy ( const uint32_t channel)
inlinestatic

Read DMA busy flag.

Parameters
channelThe index of the channel.
Note
The function passes the channel argument as an immediate, thus this must be known at compile time. As a consequence, the function must use internal linkage (static keyword) and must be always inlined. This is true also for all functions invoking this function, and passing down an argument to channel.
327 {
328#ifdef SNRT_SUPPORTS_DMA
329 uint32_t busy;
330 asm volatile("dmstati %[busy], (%[channel] << 2) | 2 \n"
331 : [ busy ] "=r"(busy)
332 : [ channel ] "i"(channel)
333 :);
334 return busy;
335#else
336 return 0;
337#endif
338}

◆ snrt_dma_disable_compute()

void snrt_dma_disable_compute ( )
inline

Disable on-the-fly compute for successive transfers.

Successive DMA transfers will be plain copies

842 {
843 snrt_dma_set_opcode(IDMA_DMOPC_OPC_PASSTHROUGH);
844}
void snrt_dma_set_opcode(uint32_t opcode)
Set a parameterless on-the-fly compute op for subsequent transfers.
Definition dma.h:816

◆ snrt_dma_disable_multicast()

void snrt_dma_disable_multicast ( )
inline

Disable multicast for successive transfers.

Successive DMA transfers will be unicast transfers

void snrt_dma_set_awuser(uint64_t field)
Set AW user field of the DMA's AXI interface.
Definition dma.h:72

◆ snrt_dma_disable_reduction()

void snrt_dma_disable_reduction ( )
inline

Disable reduction operations for successive transfers.

Successive DMA transfers will be unicast transfers

◆ snrt_dma_enable_multicast()

void snrt_dma_enable_multicast ( uint64_t mask)
inline

Enable multicast for successive transfers.

All transfers performed after this call will be multicast to all addresses specified by the address and mask pair.

Parameters
maskMulticast mask value
89 {
91 op.f.opcode = SNRT_COLLECTIVE_MULTICAST;
92 op.f.mask = mask;
94}
Definition sync_decls.h:55

◆ snrt_dma_enable_reduction()

void snrt_dma_enable_reduction ( uint64_t mask,
snrt_collective_opcode_t opcode )
inline

Enable reduction operations for successive transfers.

All transfers performed after this call will be part of a reduction involving all masters identified by the mask.

Parameters
maskMask defines all involved members
opcodeType of reduction operation
105 {
107 op.f.opcode = opcode;
108 op.f.mask = mask;
110}

◆ snrt_dma_enable_transpose()

void snrt_dma_enable_transpose ( uint32_t mode,
uint32_t tensor_m,
uint32_t tensor_n )
inline

Enable the tiled transpose of a row-major tensor for successive transfers.

Parameters
modeElement size selector; elements are 1 << mode bytes.
tensor_mRows of the source tensor, in elements.
tensor_nColumns of the source tensor, in elements.
828 {
830 IDMA_DMOPC_OPC_TRANSPOSE | ((mode & IDMA_DMOPC_RS1_TP_MODE_MASK)
831 << IDMA_DMOPC_RS1_TP_MODE_SHIFT),
832 ((tensor_m & IDMA_DMOPC_RS2_TP_TENSOR_M_MASK)
833 << IDMA_DMOPC_RS2_TP_TENSOR_M_SHIFT) |
834 ((tensor_n & IDMA_DMOPC_RS2_TP_TENSOR_N_MASK)
835 << IDMA_DMOPC_RS2_TP_TENSOR_N_SHIFT));
836}
void snrt_dma_set_opcode_params(uint32_t opcode, uint32_t params)
Set the on-the-fly compute configuration for subsequent transfers.
Definition dma.h:803

◆ snrt_dma_load_1d_tile()

snrt_dma_txid_t snrt_dma_load_1d_tile ( volatile void * dst,
volatile void * src,
size_t tile_idx,
size_t tile_size,
uint32_t prec )
inline

Load a tile of a 1D array.

Parameters
dstPointer to the tile destination.
srcPointer to the source array.
tile_idxIndex of the tile in the 1D array.
tile_sizeNumber of elements within a tile of the 1D array.
precNumber of bytes of each element in the 1D array.
476 {
477 size_t tile_nbytes = tile_size * prec;
478 return snrt_dma_start_1d(
479 (uint64_t)dst, (uint64_t)src + tile_idx * tile_nbytes, tile_nbytes);
480}
uint32_t snrt_dma_start_1d(uint64_t dst, uint64_t src, size_t size, uint32_t channel)
Start an asynchronous 1D DMA transfer with 64-bit wide pointers on a specific DMA channel.
Definition dma.h:29

◆ snrt_dma_load_1d_tile_mcast()

snrt_dma_txid_t snrt_dma_load_1d_tile_mcast ( void * dst,
void * src,
size_t tile_idx,
size_t tile_size,
uint32_t prec,
uint64_t mask )
inline

Load a tile of a 1D array.

Parameters
dstPointer to the tile destination.
srcPointer to the source array.
tile_idxIndex of the tile in the 1D array.
tile_sizeNumber of elements within a tile of the 1D array.
precNumber of bytes of each element in the 1D array.
maskMulticast mask applied on the destination address.
495 {
496 size_t tile_nbytes = tile_size * prec;
497 return snrt_dma_start_1d_mcast((uintptr_t)dst,
498 (uintptr_t)src + tile_idx * tile_nbytes,
499 tile_nbytes, mask);
500}
uint32_t snrt_dma_start_1d_mcast(uint64_t dst, uint64_t src, size_t size, uint64_t mask, uint32_t channel=0)
Start an asynchronous multicast 1D DMA transfer with 64-bit wide pointers.
Definition dma.h:167

◆ snrt_dma_load_2d_tile() [1/2]

snrt_dma_txid_t snrt_dma_load_2d_tile ( void * dst,
void * src,
size_t tile_x1_idx,
size_t tile_x0_idx,
size_t tile_x1_size,
size_t tile_x0_size,
size_t full_x0_size,
uint32_t prec )
inline

Load a 2D tile of a 2D array.

The stride in the destination tile is assumed to be that of a 1D tile, effectively. In other words, this is the same as snrt_dma_2d_to_1d().

See also
snrt_dma_load_2d_tile(void *, void *, size_t, size_t, size_t, size_t, size_t, uint32_t, size_t) for a detailed description of the parameters.
613 {
614 return snrt_dma_load_2d_tile(dst, src, tile_x1_idx, tile_x0_idx,
615 tile_x1_size, tile_x0_size, full_x0_size, prec,
616 tile_x0_size * prec);
617}
snrt_dma_txid_t snrt_dma_load_2d_tile(void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, size_t tile_ld)
Load a 2D tile of a 2D array.
Definition dma.h:582

◆ snrt_dma_load_2d_tile() [2/2]

snrt_dma_txid_t snrt_dma_load_2d_tile ( void * dst,
void * src,
size_t tile_x1_idx,
size_t tile_x0_idx,
size_t tile_x1_size,
size_t tile_x0_size,
size_t full_x0_size,
uint32_t prec,
size_t tile_ld )
inline

Load a 2D tile of a 2D array.

Parameters
dstPointer to the tile destination.
srcPointer to the source array.
tile_x1_idxOutermost coordinate of the tile in the 2D array.
tile_x0_idxInnermost coordinate of the tile in the 2D array.
tile_x1_sizeNumber of elements in the outermost dimension of the tile.
tile_x0_sizeNumber of elements in the innermost dimension of the tile.
full_x0_sizeNumber of elements in the innermost dimension of the array.
precNumber of bytes of each element in the 2D array.
tile_ldLeading dimension of the tile, in bytes.
585 {
586 size_t src_offset = 0;
587 // Advance src array in x0 and x1 dimensions, and convert to byte offset
588 src_offset += tile_x0_idx * tile_x0_size;
589 src_offset += tile_x1_idx * tile_x1_size * full_x0_size;
590 src_offset *= prec;
591 // Initiate transfer
592 return snrt_dma_start_2d((uint64_t)dst, // dst
593 (uint64_t)src + src_offset, // src
594 tile_x0_size * prec, // size
595 tile_ld, // dst_stride
596 full_x0_size * prec, // src_stride
597 tile_x1_size // repeat
598 );
599}

◆ snrt_dma_load_2d_tile_in_banks()

snrt_dma_txid_t snrt_dma_load_2d_tile_in_banks ( void * dst,
void * src,
size_t tile_x1_idx,
size_t tile_x0_idx,
size_t tile_x1_size,
size_t tile_x0_size,
size_t full_x0_size,
uint32_t prec,
size_t num_banks )
inline

Load a 2D tile of a 2D array and reshape it to occupy a subset of TCDM banks.

Parameters
dstPointer to the tile destination.
srcPointer to the source array.
tile_x1_idxOutermost coordinate of the tile in the 2D array.
tile_x0_idxInnermost coordinate of the tile in the 2D array.
tile_x1_sizeNumber of elements in the outermost dimension of the tile.
tile_x0_sizeNumber of elements in the innermost dimension of the tile.
full_x0_sizeNumber of elements in the innermost dimension of the array.
precNumber of bytes of each element in the 2D array.
num_banksNumber of banks to reshape the tile into.
703 {
704 // Calculate new tile size after reshaping the tile in the selected banks
705 size_t tile_x0_size_in_banks = (num_banks * SNRT_TCDM_BANK_WIDTH) / prec;
706 size_t tile_x1_size_in_banks =
707 ceil((tile_x1_size * tile_x0_size) / (double)tile_x0_size_in_banks);
708 size_t tile_ld = SNRT_TCDM_HYPERBANK_WIDTH;
709 return snrt_dma_load_2d_tile(dst, src, tile_x1_idx, tile_x0_idx,
710 tile_x1_size_in_banks, tile_x0_size_in_banks,
711 full_x0_size, prec, tile_ld);
712}

◆ snrt_dma_load_2d_tile_mcast() [1/3]

snrt_dma_txid_t snrt_dma_load_2d_tile_mcast ( void * dst,
void * src,
size_t tile_x1_idx,
size_t tile_x0_idx,
size_t tile_x1_size,
size_t tile_x0_size,
size_t full_x0_size,
uint32_t prec,
size_t tile_ld,
uint32_t mask )
inline

Load a 2D tile of a 2D array using multicast.

Parameters
maskMulticast mask.
See also
snrt_dma_load_2d_tile(void *, void *, size_t, size_t, size_t, size_t, size_t, uint32_t, size_t) for a description of the other parameters.
629 {
630 size_t src_offset = 0;
631 // Advance src array in x0 and x1 dimensions, and convert to byte offset
632 src_offset += tile_x0_idx * tile_x0_size;
633 src_offset += tile_x1_idx * tile_x1_size * full_x0_size;
634 src_offset *= prec;
635 // Initiate transfer
636 return snrt_dma_start_2d_mcast((uint64_t)dst, // dst
637 (uint64_t)src + src_offset, // src
638 tile_x0_size * prec, // size
639 tile_ld, // dst_stride
640 full_x0_size * prec, // src_stride
641 tile_x1_size, // repeat
642 mask // mask
643 );
644}
uint32_t snrt_dma_start_2d_mcast(uint64_t dst, uint64_t src, size_t size, size_t dst_stride, size_t src_stride, size_t repeat, uint32_t mask, uint32_t channel=0)
Start an asynchronous, multicast 2D DMA transfer with 64-bit wide pointers.
Definition dma.h:290

◆ snrt_dma_load_2d_tile_mcast() [2/3]

snrt_dma_txid_t snrt_dma_load_2d_tile_mcast ( void * dst,
void * src,
size_t tile_x1_idx,
size_t tile_x0_idx,
size_t tile_x1_size,
size_t tile_x0_size,
size_t full_x0_size,
uint32_t prec,
snrt_comm_t comm )
inline

Load a 2D tile of a 2D array using multicast.

Parameters
commCommunicator specifying which clusters to multicast to.

The stride in the destination tile is assumed to be that of a 1D tile, effectively. In other words, this is similar to snrt_dma_2d_to_1d().

See also
snrt_dma_load_2d_tile_mcast(void *, void *, size_t, size_t, size_t, size_t, size_t, uint32_t, size_t, uint32_t) for a detailed description of the parameters.
677 {
678 uint64_t mask = snrt_get_collective_mask(comm);
679 return snrt_dma_load_2d_tile_mcast(dst, src, tile_x1_idx, tile_x0_idx,
680 tile_x1_size, tile_x0_size, full_x0_size,
681 prec, mask);
682}
snrt_dma_txid_t snrt_dma_load_2d_tile_mcast(void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, size_t tile_ld, uint32_t mask)
Load a 2D tile of a 2D array using multicast.
Definition dma.h:626

◆ snrt_dma_load_2d_tile_mcast() [3/3]

snrt_dma_txid_t snrt_dma_load_2d_tile_mcast ( void * dst,
void * src,
size_t tile_x1_idx,
size_t tile_x0_idx,
size_t tile_x1_size,
size_t tile_x0_size,
size_t full_x0_size,
uint32_t prec,
uint32_t mask )
inline

Load a 2D tile of a 2D array.

The stride in the destination tile is assumed to be that of a 1D tile, effectively. In other words, this is similar to snrt_dma_2d_to_1d().

See also
snrt_dma_load_2d_tile_mcast(void *, void *, size_t, size_t, size_t, size_t, size_t, uint32_t, size_t, uint32_t) for a detailed description of the parameters.
658 {
659 return snrt_dma_load_2d_tile_mcast(dst, src, tile_x1_idx, tile_x0_idx,
660 tile_x1_size, tile_x0_size, full_x0_size,
661 prec, tile_x0_size * prec, mask);
662}

◆ snrt_dma_memset()

void snrt_dma_memset ( void * ptr,
uint8_t value,
uint32_t len )
inline

Fast memset function performed by DMA.

Parameters
ptrPointer to the start of the region.
valueValue to set.
lenNumber of bytes, must be a multiple of the DMA bus width to use the DMA.
442 {
443#ifdef SNRT_SUPPORTS_DMA
444 // We set the first 64 bytes to the value, and then we use the DMA to copy
445 // these into the remaining memory region. DMA is used only if len is
446 // larger than 64 bytes, and an integer multiple of 64 bytes.
447 size_t n_1d_transfers = len / 64;
448 size_t use_dma = (len % 64) == 0 && len > 64;
449 uint8_t *p = (uint8_t *)ptr;
450
451 uint32_t nbytes = len < 64 || !use_dma ? len : 64;
452 while (nbytes--) {
453 *p++ = value;
454 }
455
456 if (use_dma) {
457 snrt_dma_start_2d(ptr, ptr, 64, 64, 0, n_1d_transfers);
459 }
460#else
461 memset(ptr, (int)value, len);
462#endif
463}
static void snrt_dma_wait_all(const uint32_t channel=0)
Block until a specific DMA channel is idle.
Definition dma.h:394

◆ snrt_dma_reduction_load_1d_tile()

snrt_dma_txid_t snrt_dma_reduction_load_1d_tile ( void * dst,
void * src,
size_t tile_idx,
size_t tile_size,
uint32_t prec,
uint64_t mask,
snrt_collective_opcode_t opcode )
inline

Load a tile of a 1D array.

Parameters
dstPointer to the tile destination.
srcPointer to the source array.
tile_idxIndex of the tile in the 1D array.
tile_sizeNumber of elements within a tile of the 1D array.
precNumber of bytes of each element in the 1D array.
maskMask for reduction operation.
opcodeReduction operation.
514 {
515 size_t tile_nbytes = tile_size * prec;
516 return snrt_dma_start_1d_reduction((uintptr_t)dst,
517 (uintptr_t)src + tile_idx * tile_nbytes,
518 tile_nbytes, mask, opcode);
519}
uint32_t snrt_dma_start_1d_reduction(uint64_t dst, uint64_t src, size_t size, uint64_t mask, snrt_collective_opcode_t opcode, uint32_t channel=0)
Start an asynchronous reduction 1D DMA transfer with 64-bit wide pointers.
Definition dma.h:132

◆ snrt_dma_set_awuser()

void snrt_dma_set_awuser ( uint64_t field)
inline

Set AW user field of the DMA's AXI interface.

All DMA transfers performed after this call are equipped with the given AW user field

Parameters
fieldDefines the AW user field for the AXI transfer
72 {
73#ifdef SNRT_SUPPORTS_DMA
74 uint32_t user_low = (uint32_t)(field);
75 uint32_t user_high = (uint32_t)(field >> 32);
76 asm volatile("dmuser %[user_low], %[user_high] \n"
77 :
78 : [ user_low ] "r"(user_low), [ user_high ] "r"(user_high));
79#endif
80}

◆ snrt_dma_set_opcode()

void snrt_dma_set_opcode ( uint32_t opcode)
inline

Set a parameterless on-the-fly compute op for subsequent transfers.

Parameters
opcodeOpcode byte; one of the IDMA_DMOPC_OPC_* defines.
816 {
817 snrt_dma_set_opcode_params(opcode, 0u);
818}

◆ snrt_dma_set_opcode_params()

void snrt_dma_set_opcode_params ( uint32_t opcode,
uint32_t params )
inline

Set the on-the-fly compute configuration for subsequent transfers.

Parameters
opcodeOpcode word for rs1: an IDMA_DMOPC_OPC_* byte, plus the transpose element-size mode at IDMA_DMOPC_RS1_TP_MODE_SHIFT.
paramsOp-parameter word for rs2; 0 for ops that take none.

Latched at the DMOPC handshake and applied until the next DMOPC.

803 {
804#ifdef SNRT_SUPPORTS_DMA_COMPUTE
805 asm volatile("dmopc %[opcode], %[params] \n"
806 :
807 : [ opcode ] "r"(opcode), [ params ] "r"(params)
808 : "memory");
809#endif
810}

◆ snrt_dma_start_1d() [1/2]

uint32_t snrt_dma_start_1d ( uint64_t dst,
uint64_t src,
size_t size,
uint32_t channel )
inline

Start an asynchronous 1D DMA transfer with 64-bit wide pointers on a specific DMA channel.

Parameters
dstThe destination address.
srcThe source address.
sizeThe size of the transfer in bytes.
channelThe index of the channel.
Returns
The DMA transfer ID.
30 {
31#ifdef SNRT_SUPPORTS_DMA
32 uint32_t dst_lo = dst & 0xFFFFFFFF;
33 uint32_t dst_hi = dst >> 32;
34 uint32_t src_lo = src & 0xFFFFFFFF;
35 uint32_t src_hi = src >> 32;
36 uint32_t cfg = (channel << 2) | 0b00;
37 uint32_t txid;
38
39 asm volatile(
40 "dmsrc %[src_lo], %[src_hi] \n"
41 "dmdst %[dst_lo], %[dst_hi] \n"
42 "dmcpy %[txid], %[size], %[cfg] \n"
43 : [ txid ] "=r"(txid)
44 : [ src_lo ] "r"(src_lo), [ src_hi ] "r"(src_hi),
45 [ dst_lo ] "r"(dst_lo), [ dst_hi ] "r"(dst_hi), [ size ] "r"(size),
46 [ cfg ] "r"(cfg));
47 return txid;
48#else
49 memcpy((void *)dst, (const void *)src, size);
50 return 0;
51#endif
52}

◆ snrt_dma_start_1d() [2/2]

uint32_t snrt_dma_start_1d ( volatile void * dst,
volatile void * src,
size_t size,
uint32_t channel = 0 )
inline

Start an asynchronous 1D DMA transfer using native-size pointers.

This is a convenience overload of snrt_dma_start_1d(uint64_t, uint64_t, size_t, uint32_t) using void* pointers.

61 {
62 return snrt_dma_start_1d((uint64_t)dst, (uint64_t)src, size, channel);
63}

◆ snrt_dma_start_1d_mcast() [1/3]

uint32_t snrt_dma_start_1d_mcast ( uint64_t dst,
uint64_t src,
size_t size,
snrt_comm_t comm,
uint32_t channel = 0 )
inline

Start an asynchronous multicast 1D DMA transfer with 64-bit wide pointers.

Parameters
commThe communicator for the multicast operation
See also
snrt_dma_start_1d(uint64_t, uint64_t, size_t, uint32_t) for a description of the other parameters.
184 {
185 uint64_t mask = snrt_get_collective_mask(comm);
186 uint32_t txid = snrt_dma_start_1d_mcast(dst, src, size, mask, channel);
187 return txid;
188}

◆ snrt_dma_start_1d_mcast() [2/3]

uint32_t snrt_dma_start_1d_mcast ( uint64_t dst,
uint64_t src,
size_t size,
uint64_t mask,
uint32_t channel = 0 )
inline

Start an asynchronous multicast 1D DMA transfer with 64-bit wide pointers.

Parameters
maskThe mask for the multicast operation
See also
snrt_dma_start_1d(uint64_t, uint64_t, size_t, uint32_t) for a description of the other parameters.
168 {
170 uint32_t txid = snrt_dma_start_1d(dst, src, size, channel);
172 return txid;
173}
void snrt_dma_disable_multicast()
Disable multicast for successive transfers.
Definition dma.h:116
void snrt_dma_enable_multicast(uint64_t mask)
Enable multicast for successive transfers.
Definition dma.h:89

◆ snrt_dma_start_1d_mcast() [3/3]

uint32_t snrt_dma_start_1d_mcast ( volatile void * dst,
volatile void * src,
size_t size,
uint64_t mask,
uint32_t channel = 0 )
inline

Start an asynchronous multicast 1D DMA transfer using native-size pointers.

This is a convenience overload of snrt_dma_start_1d_mcast(uint64_t, uint64_t, size_t, uint64_t, uint32_t) using void* pointers.

217 {
218 return snrt_dma_start_1d_mcast((uint64_t)dst, (uint64_t)src, size, mask,
219 channel);
220}

◆ snrt_dma_start_1d_reduction() [1/3]

uint32_t snrt_dma_start_1d_reduction ( uint64_t dst,
uint64_t src,
size_t size,
snrt_comm_t comm,
snrt_collective_opcode_t opcode,
uint32_t channel = 0 )
inline

Start an asynchronous reduction 1D DMA transfer with 64-bit wide pointers.

Parameters
commThe communicator for the reduction operation
opcodeReduction operation
See also
snrt_dma_start_1d(uint64_t, uint64_t, size_t, uint32_t) for a description of the other parameters.
153 {
154 uint64_t mask = snrt_get_collective_mask(comm);
155 uint32_t txid =
156 snrt_dma_start_1d_reduction(dst, src, size, mask, opcode, channel);
157 return txid;
158}

◆ snrt_dma_start_1d_reduction() [2/3]

uint32_t snrt_dma_start_1d_reduction ( uint64_t dst,
uint64_t src,
size_t size,
uint64_t mask,
snrt_collective_opcode_t opcode,
uint32_t channel = 0 )
inline

Start an asynchronous reduction 1D DMA transfer with 64-bit wide pointers.

Parameters
maskMask defines all involved members
opcodeReduction operation
See also
snrt_dma_start_1d(uint64_t, uint64_t, size_t, uint32_t) for a description of the other parameters.
135 {
136 snrt_dma_enable_reduction(mask, opcode);
137 uint32_t txid = snrt_dma_start_1d(dst, src, size, channel);
139 return txid;
140}
void snrt_dma_disable_reduction()
Disable reduction operations for successive transfers.
Definition dma.h:122
void snrt_dma_enable_reduction(uint64_t mask, snrt_collective_opcode_t opcode)
Enable reduction operations for successive transfers.
Definition dma.h:104

◆ snrt_dma_start_1d_reduction() [3/3]

uint32_t snrt_dma_start_1d_reduction ( volatile void * dst,
volatile void * src,
size_t size,
uint64_t mask,
snrt_collective_opcode_t opcode,
uint32_t channel = 0 )
inline

Start an asynchronous reduction 1D DMA transfer using native-size pointers.

This is a convenience overload of snrt_dma_start_1d_reduction(uint64_t, uint64_t, size_t, uint64_t, uint32_t, uint32_t) using void* pointers.

202 {
203 return snrt_dma_start_1d_reduction((uint64_t)dst, (uint64_t)src, size, mask,
204 opcode, channel);
205}

◆ snrt_dma_start_2d() [1/2]

snrt_dma_txid_t snrt_dma_start_2d ( uint64_t dst,
uint64_t src,
size_t size,
size_t dst_stride,
size_t src_stride,
size_t repeat,
uint32_t channel )
inline

Start an asynchronous 2D DMA transfer with 64-bit wide pointers.

Parameters
dstThe destination address.
srcThe source address.
sizeThe size of every 1D transfer within the 2D transfer in bytes.
dst_strideThe offset between consecutive 1D transfers at the destination, in bytes.
src_strideThe offset between consecutive 1D transfers at the source, in bytes.
repeatThe number of 1D transfers composing the 2D transfer.
channelThe index of the channel.
Returns
The DMA transfer ID.
238 {
239#ifdef SNRT_SUPPORTS_DMA
240 uint32_t dst_lo = dst & 0xFFFFFFFF;
241 uint32_t dst_hi = dst >> 32;
242 uint32_t src_lo = src & 0xFFFFFFFF;
243 uint32_t src_hi = src >> 32;
244 uint32_t cfg = (channel << 2) | 0b10;
245 uint32_t txid;
246
247 asm volatile(
248 "dmsrc %[src_lo], %[src_hi] \n"
249 "dmdst %[dst_lo], %[dst_hi] \n"
250 "dmstr %[src_stride], %[dst_stride] \n"
251 "dmrep %[repeat] \n"
252 "dmcpy %[txid], %[size], %[cfg] \n"
253 : [ txid ] "=r"(txid)
254 : [ src_lo ] "r"(src_lo), [ src_hi ] "r"(src_hi),
255 [ dst_lo ] "r"(dst_lo), [ dst_hi ] "r"(dst_hi),
256 [ dst_stride ] "r"(dst_stride), [ src_stride ] "r"(src_stride),
257 [ repeat ] "r"(repeat), [ size ] "r"(size), [ cfg ] "r"(cfg));
258
259 return txid;
260#else
261 // TODO(colluca): we can implement this as a series of memcpy calls
262 return 0;
263#endif
264}

◆ snrt_dma_start_2d() [2/2]

uint32_t snrt_dma_start_2d ( volatile void * dst,
volatile void * src,
size_t size,
size_t dst_stride,
size_t src_stride,
size_t repeat,
uint32_t channel = 0 )
inline

Start an asynchronous 2D DMA transfer using native-size pointers.

This is a convenience overload of snrt_dma_start_2d(uint64_t, uint64_t, size_t, size_t, size_t, size_t, uint32_t) using void* pointers.

276 {
277 return snrt_dma_start_2d((uint64_t)dst, (uint64_t)src, size, dst_stride,
278 src_stride, repeat, channel);
279}

◆ snrt_dma_start_2d_mcast() [1/2]

uint32_t snrt_dma_start_2d_mcast ( uint64_t dst,
uint64_t src,
size_t size,
size_t dst_stride,
size_t src_stride,
size_t repeat,
uint32_t mask,
uint32_t channel = 0 )
inline

Start an asynchronous, multicast 2D DMA transfer with 64-bit wide pointers.

Parameters
maskMulticast mask.
See also
snrt_dma_start_2d(uint64_t, uint64_t, size_t, size_t, size_t, size_t, uint32_t) for a description of the other parameters.
293 {
295 uint32_t txid = snrt_dma_start_2d(dst, src, size, dst_stride, src_stride,
296 repeat, channel);
298 return txid;
299}

◆ snrt_dma_start_2d_mcast() [2/2]

uint32_t snrt_dma_start_2d_mcast ( volatile void * dst,
volatile void * src,
size_t size,
size_t dst_stride,
size_t src_stride,
size_t repeat,
uint32_t mask,
uint32_t channel = 0 )
inline

Start an asynchronous, multicast 2D DMA transfer using native-size pointers.

This is a convenience overload of snrt_dma_start_2d_mcast(uint64_t, uint64_t, size_t, size_t, size_t, size_t, uint32_t, uint32_t) using void* pointers.

312 {
313 return snrt_dma_start_2d_mcast((uint64_t)dst, (uint64_t)src, size,
314 dst_stride, src_stride, repeat, mask,
315 channel);
316}

◆ snrt_dma_start_tracking()

void snrt_dma_start_tracking ( )
inline

Start tracking of dma performance region. Does not have any implications on the HW. Only injects a marker in the DMA traces that can be analyzed.

Deprecated
417 {
418#ifdef SNRT_SUPPORTS_DMA
419 asm volatile("dmstati zero, 0 \n");
420#endif
421}

◆ snrt_dma_start_transpose() [1/2]

uint32_t snrt_dma_start_transpose ( uint64_t dst,
uint64_t src,
size_t size,
uint32_t mode,
uint32_t tensor_m,
uint32_t tensor_n,
uint32_t channel = 0 )
inline

Start an asynchronous transposing DMA transfer.

Parameters
modeElement size selector; elements are 1 << mode bytes.
tensor_mRows of the source tensor, in elements.
tensor_nColumns of the source tensor, in elements.
See also
snrt_dma_start_1d(uint64_t, uint64_t, size_t, uint32_t) for a description of the other parameters.
857 {
858 snrt_dma_enable_transpose(mode, tensor_m, tensor_n);
859 uint32_t txid = snrt_dma_start_1d(dst, src, size, channel);
861 return txid;
862}
void snrt_dma_disable_compute()
Disable on-the-fly compute for successive transfers.
Definition dma.h:842
void snrt_dma_enable_transpose(uint32_t mode, uint32_t tensor_m, uint32_t tensor_n)
Enable the tiled transpose of a row-major tensor for successive transfers.
Definition dma.h:827

◆ snrt_dma_start_transpose() [2/2]

uint32_t snrt_dma_start_transpose ( volatile void * dst,
volatile void * src,
size_t size,
uint32_t mode,
uint32_t tensor_m,
uint32_t tensor_n,
uint32_t channel = 0 )
inline

Start an asynchronous transposing DMA transfer.

See also
snrt_dma_start_transpose(uint64_t, uint64_t, size_t, uint32_t, uint32_t, uint32_t, uint32_t)
872 {
873 return snrt_dma_start_transpose((uint64_t)dst, (uint64_t)src, size, mode,
874 tensor_m, tensor_n, channel);
875}
uint32_t snrt_dma_start_transpose(uint64_t dst, uint64_t src, size_t size, uint32_t mode, uint32_t tensor_m, uint32_t tensor_n, uint32_t channel=0)
Start an asynchronous transposing DMA transfer.
Definition dma.h:854

◆ snrt_dma_stop_tracking()

void snrt_dma_stop_tracking ( )
inline

Stop tracking of dma performance region. Does not have any implications on the HW. Only injects a marker in the DMA traces that can be analyzed.

Deprecated
429 {
430#ifdef SNRT_SUPPORTS_DMA
431 asm volatile("dmstati zero, 0 \n");
432#endif
433}

◆ snrt_dma_store_1d_tile()

snrt_dma_txid_t snrt_dma_store_1d_tile ( void * dst,
void * src,
size_t tile_idx,
size_t tile_size,
uint32_t prec )
inline

Store a tile to a 1D array.

Parameters
dstPointer to the destination array.
srcPointer to the source tile.
tile_idxIndex of the tile in the 1D array.
tile_sizeNumber of elements within a tile of the 1D array.
precNumber of bytes of each element in the 1D array.
561 {
562 size_t tile_nbytes = tile_size * prec;
563 return snrt_dma_start_1d((uint64_t)dst + tile_idx * tile_nbytes,
564 (uint64_t)src, tile_nbytes);
565}

◆ snrt_dma_store_2d_tile() [1/2]

snrt_dma_txid_t snrt_dma_store_2d_tile ( void * dst,
void * src,
size_t tile_x1_idx,
size_t tile_x0_idx,
size_t tile_x1_size,
size_t tile_x0_size,
size_t full_x0_size,
uint32_t prec )
inline

Store a 2D tile of a 2D array.

The stride in the source tile is assumed to be that of a 1D tile, effectively. In other words, this is the same as snrt_dma_1d_to_2d().

See also
snrt_dma_store_2d_tile(void *, void *, size_t, size_t, size_t, size_t, size_t, uint32_t, size_t) for a detailed description of the parameters.
760 {
761 return snrt_dma_store_2d_tile(dst, src, tile_x1_idx, tile_x0_idx,
762 tile_x1_size, tile_x0_size, full_x0_size,
763 prec, tile_x0_size * prec);
764}
snrt_dma_txid_t snrt_dma_store_2d_tile(void *dst, void *src, size_t tile_x1_idx, size_t tile_x0_idx, size_t tile_x1_size, size_t tile_x0_size, size_t full_x0_size, uint32_t prec, size_t tile_ld)
Store a 2D tile to a 2D array.
Definition dma.h:729

◆ snrt_dma_store_2d_tile() [2/2]

snrt_dma_txid_t snrt_dma_store_2d_tile ( void * dst,
void * src,
size_t tile_x1_idx,
size_t tile_x0_idx,
size_t tile_x1_size,
size_t tile_x0_size,
size_t full_x0_size,
uint32_t prec,
size_t tile_ld )
inline

Store a 2D tile to a 2D array.

Parameters
dstPointer to the destination array.
srcPointer to the source tile.
tile_x1_idxOutermost coordinate of the tile in the 2D array.
tile_x0_idxInnermost coordinate of the tile in the 2D array.
tile_x1_sizeNumber of elements in the outermost dimension of the tile.
tile_x0_sizeNumber of elements in the innermost dimension of the tile.
full_x0_sizeNumber of elements in the innermost dimension of the array.
precNumber of bytes of each element in the 2D array.
tile_ldLeading dimension of the tile, in bytes.
732 {
733 size_t dst_offset = 0;
734 // Advance dst array in x0 and x1 dimensions, and convert to byte offset
735 dst_offset += tile_x0_idx * tile_x0_size;
736 dst_offset += tile_x1_idx * tile_x1_size * full_x0_size;
737 dst_offset *= prec;
738 // Initiate transfer
739 return snrt_dma_start_2d((uint64_t)dst + dst_offset, // dst
740 (uint64_t)src, // src
741 tile_x0_size * prec, // size
742 full_x0_size * prec, // dst_stride
743 tile_ld, // src_stride
744 tile_x1_size // repeat
745 );
746}

◆ snrt_dma_store_2d_tile_from_banks()

snrt_dma_txid_t snrt_dma_store_2d_tile_from_banks ( void * dst,
void * src,
size_t tile_x1_idx,
size_t tile_x0_idx,
size_t tile_x1_size,
size_t tile_x0_size,
size_t full_x0_size,
uint32_t prec,
size_t num_banks )
inline

Store a 2D tile of a 2D array from a 1D layout occupying a subset of TCDM banks.

Parameters
dstPointer to the destination array.
srcPointer to the source tile.
tile_x1_idxOutermost coordinate of the tile in the 2D array.
tile_x0_idxInnermost coordinate of the tile in the 2D array.
tile_x1_sizeNumber of elements in the outermost dimension of the tile.
tile_x0_sizeNumber of elements in the innermost dimension of the tile.
full_x0_sizeNumber of elements in the innermost dimension of the array.
precNumber of bytes of each element in the 2D array.
num_banksNumber of banks the tile is stored in.
785 {
786 // Calculate new tile size after reshaping the tile in the selected banks
787 size_t tile_x0_size_in_banks = (num_banks * SNRT_TCDM_BANK_WIDTH) / prec;
788 size_t tile_x1_size_in_banks =
789 ceil((tile_x1_size * tile_x0_size) / (double)tile_x0_size_in_banks);
790 size_t tile_ld = SNRT_TCDM_HYPERBANK_WIDTH;
791 return snrt_dma_store_2d_tile(dst, src, tile_x1_idx, tile_x0_idx,
792 tile_x1_size_in_banks, tile_x0_size_in_banks,
793 full_x0_size, prec, tile_ld);
794}

◆ snrt_dma_wait()

static void snrt_dma_wait ( snrt_dma_txid_t txid,
const uint32_t channel = 0 )
inlinestatic

Block until a DMA transfer finishes on a specific DMA channel.

Parameters
txidThe DMA transfer's ID.
channelThe index of the channel.
Note
The function passes the channel argument as an immediate, thus this must be known at compile time. As a consequence, the function must use internal linkage (static keyword) and must be always inlined. This is true also for all functions invoking this function, and passing down an argument to channel.
373 {
374#ifdef SNRT_SUPPORTS_DMA
375 asm volatile(
376 "1: \n"
377 "dmstati t0, (%[channel] << 2) | 0 \n"
378 "bltu t0, %[txid], 1b \n"
379 :
380 : [ txid ] "r"(txid), [ channel ] "i"(channel)
381 : "t0");
382#endif
383}

◆ snrt_dma_wait_all()

static void snrt_dma_wait_all ( const uint32_t channel = 0)
inlinestatic

Block until a specific DMA channel is idle.

Parameters
channelThe index of the channel.
Note
The function passes the channel argument as an immediate, thus this must be known at compile time. As a consequence, the function must use internal linkage (static keyword) and must be always inlined. This is true also for all functions invoking this function, and passing down an argument to channel.
394 {
395#ifdef SNRT_SUPPORTS_DMA
396 while (snrt_dma_busy(channel))
397 ;
398#endif
399}
static uint32_t snrt_dma_busy(const uint32_t channel)
Read DMA busy flag.
Definition dma.h:327

◆ snrt_dma_wait_all_channels()

void snrt_dma_wait_all_channels ( uint32_t num_channels)
inline

Block until the first num_channels channels are idle.

Parameters
num_channelsThe number of channels to wait on.
405 {
406 for (int c = 0; c < num_channels; c++) {
408 }
409}

◆ snrt_dma_would_block()

static uint32_t snrt_dma_would_block ( const uint32_t channel)
inlinestatic

Read DMA would_block flag.

Parameters
channelThe index of the channel.
Note
The function passes the channel argument as an immediate, thus this must be known at compile time. As a consequence, the function must use internal linkage (static keyword) and must be always inlined. This is true also for all functions invoking this function, and passing down an argument to channel.
349 {
350#ifdef SNRT_SUPPORTS_DMA
351 uint32_t would_block;
352 asm volatile("dmstati %[would_block], (%[channel] << 2) | 3 \n"
353 : [ would_block ] "=r"(would_block)
354 : [ channel ] "i"(channel)
355 :);
356 return would_block;
357#else
358 return 0;
359#endif
360}