Assuming it's the "new" SPI (or a minor revision of the one in the H7 series), it hardly matters. I defy anyone to produce a working driver based on the nonsense that passes for a theory of operation in the Reference Manual. OK, that is a slight exaggeration, but only just.
I did. And even on STM32U5 with LPTIM1 + LPGPIO + LPDMA and MSIK! Or LPBAM! Or SRD! Or whatever they called it this week!

From this autonomous perspective with chip-select automata in hardware, I get why they designed it this way.
Its been a few years since I wrote that, but I put a linked descriptor chain of 2x the length compared to the transfer size of the buffer I wanted to sample. Each transfer could only do 1 SPI frame at a time (which was 16-bits), as after a frame the SPI deasserts CS and needed to be rearmed.
So the DMA was a chain consisting of alternating RXDR reads and then a IFC write to re-arm the DMA again. To grab 64x16-bit external ADC samples, I had to have a DMA descriptor chain of 128 blocks, each taking 4 uint32's, or 2kiB of RAM..

static void _lpdmaChainGo(uint8_t ch, uint_fast8_t pos) {
assert(ch < 4);
assert(pos < MAX_CHAIN_LEN);
DMA_Channel_TypeDef* io = reinterpret_cast<DMA_Channel_TypeDef*>(LPDMA1_BASE_NS + 0x50 + 0x80*ch);
io->CTR2 = lp_chain[pos].CTR2;
io->CSAR = lp_chain[pos].SAR;
io->CDAR = lp_chain[pos].DAR;
io->CLLR = lp_chain[pos].LLR;
io->CBR1 = 2;
assert(io->CDAR != 0);
assert(io->CSAR != 0);
// GO
BitsSet(io->CCR, DMA_CCR_EN);
}
/*** LPDMA as chain ***/
struct LPDMA_MultiTransferChain_Single {
uint32_t CTR2;
uint32_t SAR;
uint32_t DAR;
uint32_t LLR;
};
#define MAX_CHAIN_LEN 128
ATTR_DATA_SRAM4 volatile LPDMA_MultiTransferChain_Single lp_chain[2*MAX_CHAIN_LEN] __attribute((aligned(16)));
ATTR_DATA_SRAM4 volatile uint16_t lp_spi_ifc;
ATTR_DATA_SRAM4 volatile uint16_t lp_tim3_arr;
static void _lpdmaChainSpiRx(uint32_t ch, uint16_t* dst, size_t length) {
assert(length <= MAX_CHAIN_LEN);
assert((length & 1) == 0);
// The chain consists of Nx2 requests:
// Each request is served in bursts.
// The first request, waits for SPI3_RX request to operate. The SPI3 is set to trigger on LPTIM3_Ch1
// After the SPI3 is triggrered, it needs to be rearmed as the EOT flag is set. This is normally a CPU activity..
// but, this is what the 2nd DMA request is for. The 2nd DMA request writes the EOT-clear word to the SPI3 peripheral
// The requests are linked in a chain of Nx2 long. The last request is configured as NULL and needs to be re-armed
// by the software IRQ. The half/full parts are handled by software. Therefore, the chain is replicated twice
// so all DAR values to the dst buffer are preloaded.
// So:
// +-----+-----+-----+-----+---------+-----+-----+-----+-----+-----+-----+---------+-----+-----+
// Chain idx | 0 | 1 | 2 | 3 | ... ... | 62 | 63 | 64 | 65 | 66 | 67 | ... ... | 126 | 127 |
// Peripheral | SPI | SPI | SPI | SPI | ... | SPI | SPI | SPI | SPI | SPI | SPI | ... | SPI | SPI |
// Register | RDR | IFC | RDR | IFC | ... | RDR | IFC | RDR | IFC | RDR | IFC | ... | RDR | IFC |
// SAR | Reg | Mem | Reg | Mem | ... | Reg | Mem | Reg | Mem | Reg | Mem | ... | Reg | Mem |
// DAR | Mem | Reg | Mem | Reg | ... | Mem | Reg | Mem | Reg | Mem | Reg | ... | Mem | Reg |
// ULL (^=update) | ^ 1 | ^ 2 | ^ 3 | ^ 4 | ... | ^63 | IRQ | ^65 | ^66 | ^67 | ^68 | ... | ^127| IRQ |
// +-----+-----+-----+-----+---------+-----+-----+-----+-----+-----+-----+---------+-----+-----+
// At idx=63 and idx=127, a software IRQ is generated to re-arm the peripheral and work on data.
// Note that IDX=0 and IDX=64 are written to the buffer for convenience, but are then copied to hardware..
assert(ch < 4);
__HAL_RCC_LPDMA1_CLK_ENABLE();
_lpdmaClr(ch);
auto tfer_size = DMA_CTR1_DDW_LOG2_0 | DMA_CTR1_SDW_LOG2_0; // 0b00=8-bit, 0b01=16-bit, 0b10=32-bit, 0b11=illegal
auto tfer_bytes = BitsAnySet(tfer_size, DMA_CTR1_DDW_LOG2_0) ? 2 :
(BitsAnySet(tfer_size, DMA_CTR1_DDW_LOG2_1) ? 4 : 1);
assert(tfer_bytes != 4);
DMA_Channel_TypeDef* io = reinterpret_cast<DMA_Channel_TypeDef*>(LPDMA1_BASE_NS + 0x50 + 0x80*ch);
BitsSet(io->CCR, DMA_CCR_TCIE/* | DMA_CCR_DTEIE | DMA_CCR_ULEIE | DMA_CCR_USEIE | DMA_CCR_SUSPIE | DMA_CCR_TOIE*/); // full transfer complete interrupts
BitsSet(io->CCR, DMA_CCR_PRIO_0 | DMA_CCR_PRIO_1); // high priority
BitsWrite(io->CTR1, DMA_CTR1_DDW_LOG2 | DMA_CTR1_SDW_LOG2, tfer_size);
BitsWrite(io->CLBAR, DMA_CLBAR_LBA, reinterpret_cast<intptr_t>(lp_chain));
BitsWrite(io->CBR1, DMA_CBR1_BNDT, tfer_bytes);
uint32_t ctr2_rdr = 0;
uint32_t ctr2_ifc = 0;
uint32_t llr_flags = 0;
// Req 2 = SPI_RX:
BitsWrite(ctr2_rdr, DMA_CTR2_REQSEL, 2);
BitsWrite(ctr2_rdr, DMA_CTR2_TCEM, DMA_CTR2_TCEM); // TCME 0b11 => transfer complete IRQ at end of channel (all LLIs processed)
// SWreq:
BitsSet(ctr2_ifc, DMA_CTR2_SWREQ);
BitsWrite(ctr2_ifc, DMA_CTR2_TCEM, DMA_CTR2_TCEM); // TCME 0b11 => transfer complete IRQ at end of channel (all LLIs processed)
// Transfer link-register, source/destination and CTR2 register
BitsSet(llr_flags, DMA_CLLR_ULL | DMA_CLLR_USA | DMA_CLLR_UDA | DMA_CLLR_UT2);
lp_spi_ifc = SPI_IFCR_EOTC | SPI_IFCR_TXTFC;
auto lli_head = lp_chain;
for (auto i = 0u; i < length; i++) {
auto lli_rdr = lli_head++;
auto lli_ifc = lli_head++;
lli_rdr->CTR2 = ctr2_rdr;
lli_ifc->CTR2 = ctr2_ifc;
lli_rdr->SAR = reinterpret_cast<intptr_t>(&(SPI3->RXDR));
lli_rdr->DAR = reinterpret_cast<intptr_t>(&(dst[i]));
lli_ifc->SAR = reinterpret_cast<intptr_t>(&(lp_spi_ifc));
lli_ifc->DAR = reinterpret_cast<intptr_t>(&(SPI3->IFCR));
lli_rdr->LLR = llr_flags | (0xFFFC & reinterpret_cast<intptr_t>(lli_ifc));
if (((i + 1) % (length / 2)) == 0) { // last word!
lli_ifc->LLR = 0;
} else {
lli_ifc->LLR = llr_flags | (0xFFFC & reinterpret_cast<intptr_t>(lli_head));
}
}
// TODO: flush dcache/SRAM4 blk to ensure it is written to SRAM4
}
void initSPI() {
// Initialize TIM3 as sampling clock
uint32_t period = 32768 / symbolrate - 1;
// LPTIM3: dma source (LPDMA1 req: 14/16)
__HAL_RCC_LPTIM3_CLK_ENABLE();
BitsSet(LPTIM3->CR, LPTIM_CR_ENABLE);
LPTIM3->ARR = period;
LPTIM3->CCR1 = 0;
BitsSet(LPTIM3->DIER, LPTIM_DIER_UEDE);
BitsClear(LPTIM3->CCMR1, LPTIM_CCMR1_CC1SEL);
BitsClear(LPTIM3->CCMR1, LPTIM_CCMR1_CC1E);
__HAL_RCC_SPI3_CLK_ENABLE();
SPI3->CR1 = 0;
SPI3->IFCR = 0xBF8; // clear all IRQs;
SPI3->CR1 = SPI_CR1_MASRX;
SPI3->CFG1 = SPI_CFG1_DSIZE_3 | // 0b01xxx = 16-bits transfers for SPI3 (limited feature set)
SPI_CFG1_RXDMAEN |
SPI_CFG1_BPASS;
SPI3->CFG2 = SPI_CFG2_AFCNTR |
SPI_CFG2_COMM_1 | // 0b10= Rx only
SPI_CFG2_MASTER |
SPI_CFG2_SSM |
SPI_CFG2_SSOE |
SPI_CFG2_AFCNTR |
(0x1 << SPI_CFG2_MSSI_Pos);
SPI3->AUTOCR = 7 << SPI_AUTOCR_TRIGSEL_Pos; // TRG7 = Lptim3_ch1
BitsSet(SPI3->AUTOCR, SPI_AUTOCR_TRIGEN);
SPI3->CR2 = 1; // burst-size
BitsSet(SPI3->CR1, SPI_CR1_SPE);
_lpdmaChainSpiRx(LPDMACH_SPI3, adcBuffer.getHwCircPtr(), adcBuffer.getHwCircSize());
_lpdmaChainGo(LPDMACH_SPI3, 0);
NVIC_EnableIRQ(LPDMA1_Channel1_IRQn);
}
But hey, stepping asides my cynical tone.. it is possible with this newer SPI peripheral! It wasn't with the old version, because as soon as SPI went active it would keep NSS asserted.. or it needs to be under software control which bypasses the whole point of using autonomous operation.
So this code can run on the U5 whilst the CPU subsystem is off. It samples an external low-power differential 12-bit ADC here at 1sps-16.384ksps. I did some measurements on the power consumption of the chip.. at a few dozen samples per second its right down at a few uA of sleep current. Iirc at 8ksps the frequent MSIK restarts did push it up to 30-40uA though.
The H5 has the same linked-list DMA peripheral, but not the SRD feature I think. We waited on a linked-list descriptor function for quite a while too... the H7 chips had it, but not on all DMA's.
Pretty much all STM32s are copy-paste on peripherals side, depending what is the latest IP version in stock. I presume they develop all their DMA, SPI, etc. IP in house, and I get it why they don't want to push too many new features at once to new silicon (avoids a PIC32MZ disaster). But honestly, chips like C5 vs H5 look very much like an economical decision of "how much MCU do you want"..