diff --git a/.reuse/dep5 b/.reuse/dep5 index 0b6cc305..a66d151f 100644 --- a/.reuse/dep5 +++ b/.reuse/dep5 @@ -23,6 +23,7 @@ Files: drivers/blk/virtio/config.json drivers/i2c/meson/config.json drivers/i2c/opentitan/config.json + drivers/network/genet/config.json drivers/network/imx/config.json drivers/network/meson/config.json drivers/network/virtio/mmio/config.json diff --git a/ci/matrix.py b/ci/matrix.py index f7ee786e..c3dca195 100644 --- a/ci/matrix.py +++ b/ci/matrix.py @@ -99,6 +99,7 @@ def listify(s: str | Sequence[str]) -> Sequence[str]: "qemu_virt_aarch64", "qemu_virt_riscv64", "rock3b", + "rpi4b_1gb", "star64", "x86_64_generic", ], diff --git a/drivers/network/genet/config.json b/drivers/network/genet/config.json new file mode 100644 index 00000000..3f393148 --- /dev/null +++ b/drivers/network/genet/config.json @@ -0,0 +1,32 @@ +{ + "compatible": [ + "brcm,bcm2711-genet-v5" + ], + "resources": { + "regions": [ + { + "name": "regs", + "perms": "rw", + "size": 65536, + "dt_index": 0 + }, + { + "name": "device_rx_ring", + "size": 32768 + }, + { + "name": "device_tx_ring", + "size": 32768 + }, + { + "name": "mbox_message", + "size": 32768 + } + ], + "irqs": [ + { + "dt_index": 0 + } + ] + } +} diff --git a/drivers/network/genet/eth_driver.mk b/drivers/network/genet/eth_driver.mk new file mode 100644 index 00000000..a85ac7b7 --- /dev/null +++ b/drivers/network/genet/eth_driver.mk @@ -0,0 +1,27 @@ +# +# Copyright 2024, UNSW +# +# SPDX-License-Identifier: BSD-2-Clause +# +# Include this snippet in your project Makefile to build +# the GENET NIC driver +# +# NOTES +# Generates eth_driver.elf (alternative unique name eth_driver_genet.elf) +# Expects libsddf_util_debug.a to be in LIBS + +ETHERNET_DRIVER_DIR := $(dir $(lastword $(MAKEFILE_LIST))) +CHECK_NETDRV_GENET_FLAGS_MD5:=.netdrv_genet_cflags-$(shell echo -- ${CFLAGS} ${CFLAGS_network} | shasum | sed 's/ *-//') + +${CHECK_NETDRV_GENET_FLAGS_MD5}: + -rm -f .netdrv_genet_cflags-* + touch $@ + +eth_driver_genet.elf: network/genet/ethernet.o libsddf_util.a + $(LD) $(LDFLAGS) $^ $(LIBS) -o $@ + +network/genet/ethernet.o: ${ETHERNET_DRIVER_DIR}/ethernet.c ${CHECK_NETDRV_FLAGS_MD5} | $(SDDF_LIBC_INCLUDE) + mkdir -p network/genet + ${CC} -c ${CFLAGS} ${CFLAGS_network} -I ${ETHERNET_DRIVER_DIR} -o $@ $< + +-include genet/ethernet.d diff --git a/drivers/network/genet/ethernet.c b/drivers/network/genet/ethernet.c new file mode 100644 index 00000000..80607685 --- /dev/null +++ b/drivers/network/genet/ethernet.c @@ -0,0 +1,543 @@ +/* + * Copyright 2026, UNSW + * SPDX-License-Identifier: BSD-2-Clause + * + * This driver is based on the Linux driver: + * drivers/net/ethernet/broadcom/genet/bcmgenet.c + * which is: Copyright (c) 2014-2017 Broadcom + * + * Also referred to: + * https://github.com/u-boot/u-boot/blob/master/drivers/net/bcmgenet.c + * https://github.com/RT-Thread/rt-thread/blob/master/bsp/raspberry-pi/raspi4-32/driver/drv_eth.c + * BCM54213PE Datasheet + */ + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "ethernet.h" + +__attribute__((__section__(".device_resources"))) device_resources_t device_resources; +__attribute__((__section__(".net_driver_config"))) net_driver_config_t config; +__attribute__((__section__(".timer_client_config"))) timer_client_config_t timer_config; + +volatile struct genet_regs *eth; +volatile struct mbox_regs *mbox_regs; +volatile uint32_t *mbox; + +volatile struct genet_dma_ring_rx *ring_rx; +volatile struct genet_dma_ring_tx *ring_tx; + +/* HW ring buffer data type */ +typedef struct { + uint32_t tail; /* index to insert at */ + uint32_t head; /* index to remove from */ + uint32_t capacity; /* capacity of description ring */ + uint32_t desc_id_mask; /* mask of description id */ + uint32_t index_mask; /* mask of index in ring */ + volatile struct genet_dma_desc *descr; /* buffer descriptor array */ +} hw_ring_t; + +hw_ring_t rx; +hw_ring_t tx; + +net_queue_handle_t rx_queue; +net_queue_handle_t tx_queue; + +static inline bool hw_ring_full(hw_ring_t *ring) +{ + return ring->tail - ring->head == ring->capacity; +} + +static inline bool hw_ring_empty(hw_ring_t *ring) +{ + return ring->tail - ring->head == 0; +} + +static void update_ring_slot(hw_ring_t *ring, unsigned int idx, uintptr_t phys, uint32_t stat) +{ + volatile struct genet_dma_desc *d = &(ring->descr[idx]); + d->addr_lo = phys & 0xFFFFFFFF; + d->addr_hi = phys >> 32; + d->status = stat; +} + +static void sleep_us(uint32_t us) +{ + uint64_t start = sddf_timer_time_now(timer_config.driver_id); + while ((sddf_timer_time_now(timer_config.driver_id) - start) < us * NS_IN_US); +} + +static void bcmgenet_mdio_write(uint8_t reg_addr, uint16_t val) +{ + uint32_t cmd = MDIO_WR | (GENET_PHY_ID << MDIO_PMD_SHIFT) | ((reg_addr & MDIO_REG_MASK) << MDIO_REG_SHIFT) | val; + eth->umac_mdio_cmd = cmd; + + uint32_t reg = eth->umac_mdio_cmd | MDIO_START_BUSY; + eth->umac_mdio_cmd = reg; + + while (eth->umac_mdio_cmd & MDIO_START_BUSY); +} + +static uint16_t bcmgenet_mdio_read(uint8_t reg_addr) +{ + uint32_t cmd = MDIO_RD | (GENET_PHY_ID << MDIO_PMD_SHIFT) | ((reg_addr & MDIO_REG_MASK) << MDIO_REG_SHIFT); + eth->umac_mdio_cmd = cmd; + + uint32_t reg = eth->umac_mdio_cmd | MDIO_START_BUSY; + eth->umac_mdio_cmd = reg; + + while (eth->umac_mdio_cmd & MDIO_START_BUSY); + + return eth->umac_mdio_cmd & 0xFFFF; +} + +static void rx_provide(void) +{ + bool reprocess = true; + while (reprocess) { + while (!hw_ring_full(&rx) && !net_queue_empty_free(&rx_queue)) { + net_buff_desc_t buffer; + int err = net_dequeue_free(&rx_queue, &buffer); + assert(!err); + + uint32_t idx = rx.tail & rx.desc_id_mask; + // The NIC uses the first CHECKSUM_TSB_LENGTH bytes for checksum + // offload. Since our Rx virtualiser expects the ethernet header to + // start at byte 0, we let the NIC write the checksum offload data + // to the earlier bytes + update_ring_slot(&rx, idx, buffer.io_or_offset - CHECKSUM_TSB_LENGTH, 0); + rx.tail++; + } + THREAD_MEMORY_RELEASE(); + // Doorbell the device + ring_rx->cons_index = (rx.tail - NUM_DESCS) & rx.index_mask; + + // Only request a notification from virtualiser if HW ring not full + if (!hw_ring_full(&rx)) { + net_request_signal_free(&rx_queue); + } else { + net_cancel_signal_free(&rx_queue); + } + reprocess = false; + + if (!net_queue_empty_free(&rx_queue) && !hw_ring_full(&rx)) { + net_cancel_signal_free(&rx_queue); + reprocess = true; + } + } +} + +static void rx_return(void) +{ + bool packets_transferred = false; + + // Each register read takes over 300 cycles, so we read prod_index once + // for optimisation. The packets arrived before clearing IRQ status will + // be handled in next iteration. + uint32_t prod_index = ring_rx->prod_index & rx.index_mask; + THREAD_MEMORY_ACQUIRE(); + while (!hw_ring_empty(&rx)) { + if ((rx.head & rx.index_mask) == prod_index) { + break; + } + uint32_t idx = rx.head & rx.desc_id_mask; + volatile struct genet_dma_desc *d = &(rx.descr[idx]); + + uint64_t addr = ((uint64_t)(d->addr_hi) << 32) | d->addr_lo; + // Return the buffer address to its previous value + net_buff_desc_t buffer = { addr + CHECKSUM_TSB_LENGTH, d->status >> DMA_BUFLENGTH_SHIFT }; + int err = net_enqueue_active(&rx_queue, buffer); + assert(!err); + + packets_transferred = true; + rx.head++; + } + + if (packets_transferred && net_require_signal_active(&rx_queue)) { + net_cancel_signal_active(&rx_queue); + sddf_notify(config.virt_rx.id); + } +} + +static void tx_provide() +{ + bool reprocess = true; + while (reprocess) { + while (!(hw_ring_full(&tx)) && !net_queue_empty_active(&tx_queue)) { + net_buff_desc_t buffer; + int err = net_dequeue_active(&tx_queue, &buffer); + assert(!err); + + uint32_t idx = tx.tail & tx.desc_id_mask; + uint32_t stat = (buffer.len << DMA_BUFLENGTH_SHIFT) | (0x3F << DMA_TX_QTAG_SHIFT) | DMA_TX_APPEND_CRC + | DMA_TX_DO_CSUM | DMA_SOP | DMA_EOP; + update_ring_slot(&tx, idx, buffer.io_or_offset, stat); + + tx.tail++; + } + THREAD_MEMORY_RELEASE(); + ring_tx->prod_index = tx.tail & tx.index_mask; + + net_request_signal_active(&tx_queue); + reprocess = false; + + if (!hw_ring_full(&tx) && !net_queue_empty_active(&tx_queue)) { + net_cancel_signal_active(&tx_queue); + reprocess = true; + } + } +} + +static void tx_return(void) +{ + bool enqueued = false; + uint32_t cons_index = ring_tx->cons_index & tx.index_mask; + THREAD_MEMORY_ACQUIRE(); + while (!hw_ring_empty(&tx)) { + if ((tx.head & tx.index_mask) == cons_index) { + break; + } + uint32_t idx = tx.head & tx.desc_id_mask; + volatile struct genet_dma_desc *d = &(tx.descr[idx]); + + uint64_t addr = ((uint64_t)(d->addr_hi) << 32) | d->addr_lo; + net_buff_desc_t buffer = { addr, 0 }; + int err = net_enqueue_free(&tx_queue, buffer); + assert(!err); + + enqueued = true; + tx.head++; + } + + if (enqueued && net_require_signal_free(&tx_queue)) { + net_cancel_signal_free(&tx_queue); + sddf_notify(config.virt_tx.id); + } +} + +static void handle_irq(void) +{ + uint32_t irq_status = eth->intrl2_0_cpu_stat & ~(eth->intrl2_0_cpu_stat_mask); + eth->intrl2_0_cpu_clear = irq_status; + while (irq_status) { + if (irq_status & GENET_IRQ_TXDMA_DONE) { + tx_return(); + tx_provide(); + } + if (irq_status & GENET_IRQ_RXDMA_DONE) { + rx_return(); + rx_provide(); + } + irq_status = eth->intrl2_0_cpu_stat & ~(eth->intrl2_0_cpu_stat_mask); + eth->intrl2_0_cpu_clear = irq_status; + } +} + +static void eth_setup(void) +{ + eth = device_resources.regions[0].region.vaddr; + + uint8_t version_major = (eth->sys_rev_ctrl >> 24) & 0x0f; + if (version_major != 6) { + sddf_dprintf("Unsupported GENET version\n"); + } + + // Set PHY interface + eth->sys_port_ctrl = PORT_MODE_EXT_GPHY; + + // Rbuf clear + eth->sys_rbuf_flush_ctrl = 0; + + // Disable MAC while updating its registers + eth->umac_cmd = 0; + // issue soft reset with (rg)mii loopback to ensure a stable rxclk + eth->umac_cmd = CMD_SW_RESET | CMD_LCL_LOOP_EN; + + // MDIO init + uint32_t uid_high = bcmgenet_mdio_read(BCM54213PE_PHY_IDENTIFIER_HIGH); + uint32_t uid_low = bcmgenet_mdio_read(BCM54213PE_PHY_IDENTIFIER_LOW); + if (((uid_high << 16) | (uid_low & 0xFFFF)) == 0) { + sddf_dprintf("ERROR: invalid ethernet UID '0'\n"); + return; + } + + // reset phy + bcmgenet_mdio_write(BCM54213PE_MII_CONTROL, MII_CONTROL_PHY_RESET); + // read control reg + bcmgenet_mdio_read(BCM54213PE_MII_CONTROL); + // reset phy again + bcmgenet_mdio_write(BCM54213PE_MII_CONTROL, MII_CONTROL_PHY_RESET); + // read control reg + bcmgenet_mdio_read(BCM54213PE_MII_CONTROL); + // read status reg + bcmgenet_mdio_read(BCM54213PE_MII_STATUS); + // read status reg + bcmgenet_mdio_read(BCM54213PE_IEEE_EXTENDED_STATUS); + bcmgenet_mdio_read(BCM54213PE_AUTO_NEGOTIATION_ADV); + + bcmgenet_mdio_read(BCM54213PE_MII_STATUS); + bcmgenet_mdio_read(BCM54213PE_CONTROL); + // half full duplex capability + bcmgenet_mdio_write(BCM54213PE_CONTROL, (CONTROL_HALF_DUPLEX_CAPABILITY | CONTROL_FULL_DUPLEX_CAPABILITY)); + bcmgenet_mdio_read(BCM54213PE_MII_CONTROL); + + // set mii control + bcmgenet_mdio_write(BCM54213PE_MII_CONTROL, + (MII_CONTROL_AUTO_NEGOTIATION_ENABLED | MII_CONTROL_AUTO_NEGOTIATION_RESTART + | MII_CONTROL_PHY_FULL_DUPLEX | MII_CONTROL_SPEED_SELECTION)); + + while (~bcmgenet_mdio_read(BCM54213PE_MII_STATUS) & MII_STATUS_AUTO_NEGOTIATION_COMPLETE); + + // Set MAC address + mbox[0] = 8 * 4; // length of the message + mbox[1] = MBOX_REQUEST; // this is a request message + mbox[2] = MBOX_TAG_HARDWARE_GET_MAC_ADDRESS; + mbox[3] = 6; // buffer size + mbox[4] = 0; // len + mbox[5] = 0; + mbox[6] = 0; + mbox[7] = MBOX_TAG_LAST; + + // 0x8 is the channel identifier for Property Tags + unsigned int r = device_resources.regions[3].io_addr | 0x8; + /* wait until we can write to the mailbox */ + while (mbox_regs->status & MBOX_FULL); + // write the address of our message to the mailbox with channel identifier + mbox_regs->write = r; + // wait for the response + while (1) { + // check if response is ready + while (mbox_regs->status & MBOX_EMPTY); + // check if response is for us + if (r == mbox_regs->read) { + if (mbox[1] != MBOX_RESPONSE) { + sddf_dprintf("ERROR: Invalid mbox response\n"); + return; + } + break; + } + } + char *mac = (char *)&mbox[5]; + eth->umac_mac0 = mac[0] << 24 | mac[1] << 16 | mac[2] << 8 | mac[3]; + eth->umac_mac1 = mac[4] << 8 | mac[5]; + + // UMAC Reset + eth->sys_rbuf_flush_ctrl |= BIT(1); + eth->sys_rbuf_flush_ctrl &= ~BIT(1); + sleep_us(10); + + eth->sys_rbuf_flush_ctrl = 0; + sleep_us(10); + + eth->umac_cmd = 0; + eth->umac_cmd = CMD_SW_RESET | CMD_LCL_LOOP_EN; + sleep_us(2); + + eth->umac_cmd = 0; + eth->umac_mib_ctrl = MIB_RESET_RX | MIB_RESET_TX | MIB_RESET_RUNT; + eth->umac_mib_ctrl = 0; + eth->umac_max_frame_len = ENET_MAX_MTU_SIZE; + + // Disable this bit to not pad two bytes at the beginning of every packet + eth->rbuf_ctrl &= ~RBUF_ALIGN_2B; + eth->rbuf_tbuf_size_ctrl = 1; + + // Disable DMA + eth->dma_tx.ctrl &= ~BIT(DMA_EN); + eth->dma_rx.ctrl &= ~BIT(DMA_EN); + eth->umac_tx_flush = 1; + sleep_us(100); + eth->umac_tx_flush = 0; + + // Rx Ring Init + ring_rx = (struct genet_dma_ring_rx *)ð->dma_rx.ring; + eth->dma_rx.burst_size = DMA_MAX_BURST_LENGTH; + ring_rx->start_addr = 0; + ring_rx->read_ptr = 0; + ring_rx->write_ptr = 0; + ring_rx->end_addr = NUM_DESCS * DESC_SIZE / 4 - 1; + ring_rx->prod_index = 0; + ring_rx->cons_index = 0; + ring_rx->buf_size = (NUM_DESCS << 16) | NET_BUFFER_SIZE; + ring_rx->mbuf_done_thresh = 0x1; + ring_rx->xon_xoff_thresh = (NUM_DESCS >> 4) | (5 << 16); + // We only use the default ring (i.e. ring 16) + eth->dma_rx.ring_cfg = BIT(DEFAULT_Q); + eth->rbuf_ctrl |= RBUF_64B_EN; + + // Tx Ring Init + ring_tx = (struct genet_dma_ring_tx *)ð->dma_tx.ring; + eth->dma_tx.burst_size = DMA_MAX_BURST_LENGTH; + ring_tx->start_addr = 0; + ring_tx->read_ptr = 0; + ring_tx->write_ptr = 0; + ring_tx->prod_index = ring_tx->cons_index; + ring_tx->end_addr = NUM_DESCS * DESC_SIZE / 4 - 1; + ring_tx->mbuf_done_thresh = 0x80; + ring_tx->flow_period = 0; + ring_tx->buf_size = (NUM_DESCS << 16) | NET_BUFFER_SIZE; + eth->dma_tx.ring_cfg = BIT(DEFAULT_Q); + eth->tbuf_ctrl |= TBUF_64B_EN; + // No timeout for Tx Coalescing but IRQs generated after mbuf_done_thresh or empty buffer + + // Enable DMA + uint32_t dma_ctrl = (1 << (DEFAULT_Q + DMA_RING_BUF_EN_SHIFT)) | DMA_EN; + eth->dma_tx.ctrl = dma_ctrl; + eth->dma_rx.ctrl |= dma_ctrl; + + // Adjust Link + uint32_t oob_ctrl = eth->ext_rgmii_oob_ctrl | RGMII_LINK | RGMII_MODE_EN | ID_MODE_DIS; + eth->ext_rgmii_oob_ctrl = oob_ctrl; + sleep_us(1000); + eth->umac_cmd = UMAC_SPEED_1000 << CMD_SPEED_SHIFT; + + // Index Reset + tx.descr = (struct genet_dma_desc *)ð->dma_tx.descs; + tx.head = ring_tx->cons_index; + tx.tail = ring_tx->prod_index; + tx.capacity = NUM_DESCS; + tx.desc_id_mask = NUM_DESCS - 1; + tx.index_mask = 0xFFFF; + + rx.head = ring_rx->prod_index; + rx.tail = rx.head; + rx.capacity = NUM_DESCS; + rx.desc_id_mask = NUM_DESCS - 1; + rx.index_mask = 0xFFFF; + rx.descr = (struct genet_dma_desc *)ð->dma_rx.descs; + + // Since we assume that the first CHECKSUM_TSB_LENGTH (64) bytes before each + // buffer can be used as a checksum scratchpad for the NIC, we cannot use + // the first sDDF buffer Fill empty buffers for Rx. So we dequeue it and + // drop it at init time + net_buff_desc_t buffer; + int err = net_dequeue_free(&rx_queue, &buffer); + assert(!err); + + // Fill empty buffers for Rx + rx_provide(); + + ring_rx->cons_index = rx.head; + ring_rx->prod_index = rx.head; + + // Enable promisc mode + eth->umac_cmd = eth->umac_cmd | CMD_PROMISC; + eth->umac_mdf_ctrl = 0; + + // Enable Rx/Tx + eth->umac_cmd |= (CMD_TX_EN | CMD_RX_EN); + + // Clear IRQ status + uint32_t irq_status = eth->intrl2_0_cpu_stat & ~(eth->intrl2_1_cpu_stat_mask); + eth->intrl2_0_cpu_clear = irq_status; + + // Enable IRQ + eth->intrl2_0_cpu_clear_mask = GENET_IRQ_TXDMA_DONE | GENET_IRQ_RXDMA_DONE; +} + +uint32_t rpi4_get_cpu_frequency() +{ + mbox[0] = 8 * 4; // Length of the message + mbox[1] = MBOX_REQUEST; // Request code + mbox[2] = MBOX_TAG_HARDWARE_GET_CLK_RATE; // tag + mbox[3] = 8; // Buffer size + mbox[4] = 0; // Response size + mbox[5] = 0x00000003; // Clock ID: ARM + mbox[6] = MBOX_TAG_LAST; // End tag + + // 0x8 is the channel identifier for Property Tags + unsigned int r = device_resources.regions[3].io_addr | 0x8; + /* wait until we can write to the mailbox */ + while (mbox_regs->status & MBOX_FULL); + // write the address of our message to the mailbox with channel identifier + mbox_regs->write = r; + // wait for the response + while (1) { + // check if response is ready + while (mbox_regs->status & MBOX_EMPTY); + // check if response is for us + if (r == mbox_regs->read) { + if (mbox[1] != MBOX_RESPONSE) { + return 0; + } + break; + } + } + + uint32_t *cpu_freq = (uint32_t *)&mbox[6]; + return *cpu_freq; +} + +void rpi4_set_cpu_frequency(uint32_t freq) +{ + mbox[0] = 8 * 4; // Length of the message + mbox[1] = MBOX_REQUEST; // Request code + mbox[2] = MBOX_TAG_HARDWARE_SET_CLK_RATE; // tag + mbox[3] = 12; // Buffer size + mbox[4] = 8; // Response size + mbox[5] = 0x00000003; // Clock ID: ARM + mbox[6] = freq; // Frequency in Hz + mbox[7] = MBOX_TAG_LAST; // End tag + + // 0x8 is the channel identifier for Property Tags + unsigned int r = device_resources.regions[3].io_addr | 0x8; + /* wait until we can write to the mailbox */ + while (mbox_regs->status & MBOX_FULL); + // write the address of our message to the mailbox with channel identifier + mbox_regs->write = r; + // wait for the response + while (1) { + // check if response is ready + while (mbox_regs->status & MBOX_EMPTY); + // check if response is for us + if (r == mbox_regs->read) { + if (mbox[1] != MBOX_RESPONSE) { + return; + } + break; + } + } +} + +void init(void) +{ + mbox_regs = (struct mbox_regs *)0x3000880; + mbox = device_resources.regions[3].region.vaddr; + + net_queue_init(&rx_queue, config.virt_rx.free_queue.vaddr, config.virt_rx.active_queue.vaddr, + config.virt_rx.num_buffers); + net_queue_init(&tx_queue, config.virt_tx.free_queue.vaddr, config.virt_tx.active_queue.vaddr, + config.virt_tx.num_buffers); + + rpi4_set_cpu_frequency(1000000000); + + eth_setup(); + + tx_provide(); +} + +void notified(sddf_channel ch) +{ + if (ch == device_resources.irqs[0].id) { + handle_irq(); + + sddf_deferred_irq_ack(ch); + } else if (ch == config.virt_rx.id) { + rx_provide(); + } else if (ch == config.virt_tx.id) { + tx_provide(); + } else { + sddf_dprintf("ETH|LOG: received notification on unexpected channel: %u\n", ch); + } +} diff --git a/drivers/network/genet/ethernet.h b/drivers/network/genet/ethernet.h new file mode 100644 index 00000000..45f5269a --- /dev/null +++ b/drivers/network/genet/ethernet.h @@ -0,0 +1,292 @@ +/* + * Copyright 2026, UNSW + * SPDX-License-Identifier: BSD-2-Clause + * + * This driver is based on the Linux driver: + * drivers/net/ethernet/broadcom/genet/bcmgenet.c + * which is: Copyright (c) 2014-2017 Broadcom + * + * Also referred to: + * https://github.com/u-boot/u-boot/blob/master/drivers/net/bcmgenet.c + * https://github.com/RT-Thread/rt-thread/blob/master/bsp/raspberry-pi/raspi4-32/driver/drv_eth.c + * BCM54213PE Datasheet + */ + +#pragma once + +#define PORT_MODE_EXT_GPHY 3 +#define CMD_PROMISC BIT(4) +#define CMD_SW_RESET BIT(13) +#define CMD_LCL_LOOP_EN BIT(15) + +#define GENET_PHY_ID 1 +#define MDIO_START_BUSY BIT(29) +#define MDIO_READ_FAIL BIT(28) +#define MDIO_RD (2 << 26) +#define MDIO_WR BIT(26) +#define MDIO_PMD_SHIFT (21) +#define MDIO_PMD_MASK (0x1f) +#define MDIO_REG_SHIFT (16) +#define MDIO_REG_MASK (0x1f) + +#define CMD_TX_EN BIT(0) +#define CMD_RX_EN BIT(1) +#define UMAC_SPEED_10 (0) +#define UMAC_SPEED_100 (1) +#define UMAC_SPEED_1000 (2) +#define UMAC_SPEED_2500 (3) +#define CMD_SPEED_SHIFT (2) +#define CMD_SPEED_MASK (3) +#define CMD_SW_RESET BIT(13) +#define CMD_LCL_LOOP_EN BIT(15) + +#define GENET_IRQ_TXDMA_DONE BIT(16) +#define GENET_IRQ_RXDMA_DONE BIT(13) + +#define RGMII_LINK BIT(4) +#define RGMII_MODE_EN BIT(6) +#define ID_MODE_DIS BIT(16) + +#define MIB_RESET_RX BIT(0) +#define MIB_RESET_RUNT BIT(1) +#define MIB_RESET_TX BIT(2) + +/* Body(1500) + EH_SIZE(14) + VLANTAG(4) + BRCMTAG(6) + FCS(4) = 1528. + * MTU must be a multiple of 256, so we set ENET_PAD to 8. + * RSB/TSB (64Bytes) is not included. + */ +#define ETH_DATA_LEN (1500) +#define ETH_HLEN (14) +#define VLAN_HLEN (4) +#define ETH_FCS_LEN (4) +#define ENET_BRCM_TAG_LEN (6) +#define ENET_PAD (8) +#define ENET_MAX_MTU_SIZE (ETH_DATA_LEN + ETH_HLEN + \ + VLAN_HLEN + ENET_BRCM_TAG_LEN + \ + ETH_FCS_LEN + ENET_PAD) + +#define GENET_RBUF_OFF (0x0300) +#define RBUF_TBUF_SIZE_CTRL (GENET_RBUF_OFF + 0xb4) +#define RBUF_CTRL (GENET_RBUF_OFF + 0x00) +#define RBUF_ALIGN_2B BIT(1) + +#define RBUF_64B_EN BIT(0) +#define RBUF_CTRL_CHKSUM_EN BIT(0) +#define TBUF_64B_EN BIT(0) +#define TBUF_CTRL_CHKSUM_EN BIT(0) + +/* Tx/Rx Dma Descriptor common bits */ +#define DMA_EN BIT(0) +#define DMA_RING_BUF_EN_SHIFT (0x01) +#define DMA_RING_BUF_EN_MASK (0xffff) +#define DMA_BUFLENGTH_MASK (0x0fff) +#define DMA_BUFLENGTH_SHIFT (16) +#define DMA_RING_SIZE_SHIFT (16) +#define DMA_OWN (0x8000) +#define DMA_EOP (0x4000) +#define DMA_SOP (0x2000) +#define DMA_WRAP (0x1000) +#define DMA_MAX_BURST_LENGTH (0x8) +/* Tx specific DMA descriptor bits */ +#define DMA_TX_UNDERRUN (0x0200) +#define DMA_TX_APPEND_CRC (0x0040) +#define DMA_TX_OW_CRC (0x0020) +#define DMA_TX_DO_CSUM (0x0010) +#define DMA_TX_CHKSUM_EN BIT(15) +#define DMA_TX_QTAG_SHIFT (7) + +#define DMA_TIMEOUT_MASK 0xFFFF + +#define DEFAULT_Q 0x10 +#define RING_INDEX_CAPACITY 0x10000 + +#define BCM54213PE_MII_CONTROL (0x00) +#define BCM54213PE_MII_STATUS (0x01) +#define BCM54213PE_PHY_IDENTIFIER_HIGH (0x02) +#define BCM54213PE_PHY_IDENTIFIER_LOW (0x03) + +#define BCM54213PE_AUTO_NEGOTIATION_ADV (0x04) +#define BCM54213PE_AUTO_NEGOTIATION_LINK (0x05) +#define BCM54213PE_AUTO_NEGOTIATION_EXPANSION (0x06) + +#define BCM54213PE_NEXT_PAGE_TX (0x07) + +#define BCM54213PE_PARTNER_RX (0x08) + +#define BCM54213PE_CONTROL (0x09) +#define BCM54213PE_STATUS (0x0A) + +#define BCM54213PE_IEEE_EXTENDED_STATUS (0x0F) +#define BCM54213PE_PHY_EXTENDED_CONTROL (0x10) +#define BCM54213PE_PHY_EXTENDED_STATUS (0x11) + +#define BCM54213PE_RECEIVE_ERROR_COUNTER (0x12) +#define BCM54213PE_FALSE_C_S_COUNTER (0x13) +#define BCM54213PE_RECEIVE_NOT_OK_COUNTER (0x14) + +#define BCM54213PE_VERSION_B1 (0x600d84a2) +#define BCM54213PE_VERSION_X (0x600d84a0) + +//BCM54213PE_MII_CONTROL +#define MII_CONTROL_PHY_RESET BIT(15) +#define MII_CONTROL_AUTO_NEGOTIATION_ENABLED BIT(12) +#define MII_CONTROL_AUTO_NEGOTIATION_RESTART BIT(9) +#define MII_CONTROL_PHY_FULL_DUPLEX BIT(8) +#define MII_CONTROL_SPEED_SELECTION BIT(6) + +//BCM54213PE_MII_STATUS +#define MII_STATUS_LINK_UP BIT(2) +#define MII_STATUS_AUTO_NEGOTIATION_COMPLETE BIT(5) + +//BCM54213PE_CONTROL +#define CONTROL_FULL_DUPLEX_CAPABILITY BIT(9) +#define CONTROL_HALF_DUPLEX_CAPABILITY BIT(8) + +struct genet_dma_desc { + uint32_t status; + uint32_t addr_lo; + uint32_t addr_hi; +}; + +#define NUM_DESCS 256 +#define DESC_SIZE sizeof(struct genet_dma_desc) + +struct genet_dma_ring_rx { + uint32_t write_ptr; // 0x00 + uint32_t unused1; // 0x04 + uint32_t prod_index; // 0x08 + uint32_t cons_index; // 0x0C + uint32_t buf_size; // 0x10 + uint32_t start_addr; // 0x14 + uint32_t unused2; // 0x18 + uint32_t end_addr; // 0x1C + uint32_t unused3; // 0x20 + uint32_t mbuf_done_thresh; // 0x24 + uint32_t xon_xoff_thresh; // 0x28 + uint32_t read_ptr; // 0x2C + uint8_t unused4[16]; // 0x30-0x40 +}; + +struct genet_dma_ring_tx { + uint32_t read_ptr; // 0x00 + uint32_t unused1; // 0x04 + uint32_t cons_index; // 0x08 + uint32_t prod_index; // 0x0C + uint32_t buf_size; // 0x10 + uint32_t start_addr; // 0x14 + uint32_t unused2; // 0x18 + uint32_t end_addr; // 0x1C + uint32_t unused3; // 0x20 + uint32_t mbuf_done_thresh; // 0x24 + uint32_t flow_period; // 0x28 + uint32_t write_ptr; // 0x2C + uint8_t unused4[16]; // 0x30-0x40 +}; + +typedef struct genet_dma_ring { + uint8_t x[64]; +} genet_dma_ring_t; + +struct genet_dma { + struct genet_dma_desc descs[NUM_DESCS]; // 0x000 + genet_dma_ring_t default_rings[DEFAULT_Q]; // 0xC00-0x1000 + genet_dma_ring_t ring; // 0x1000-0x1040 + uint32_t ring_cfg; // 0x1040 + uint32_t ctrl; // 0x1044 + uint8_t unused1[4]; // 0x1048 + uint32_t burst_size; // 0x104C + uint8_t unused3[92]; // 0x1050-0x10AC + uint32_t ring16_timeout; // 0x10AC +}; + +struct genet_regs { + uint32_t sys_rev_ctrl; // 0x00 + uint32_t sys_port_ctrl; // 0x04 + uint32_t sys_rbuf_flush_ctrl; // 0x08 + uint32_t sys_tbuf_flush_ctrl; // 0x0C + uint8_t sys_unused[112]; // 0x10-0x80 + uint32_t ext_pwr_mgmt; // 0x80 + uint8_t ext_unused1[8]; // 0x84-0x8C + uint32_t ext_rgmii_oob_ctrl; // 0x8C + uint8_t ext_unused2[12]; // 0x90-0x9C + uint32_t ext_gphy_ctrl; // 0x9C + uint8_t ext_unused3[352]; // 0xA0-0x200 + uint32_t intrl2_0_cpu_stat; // 0x200 + uint32_t intrl2_0_cpu_set; // 0x204 + uint32_t intrl2_0_cpu_clear; // 0x208 + uint32_t intrl2_0_cpu_stat_mask; // 0x20C + uint32_t intrl2_0_cpu_set_mask; // 0x210 + uint32_t intrl2_0_cpu_clear_mask; // 0x214 + uint8_t intrl2_0_unused2[40]; // 0x218-0x240 + uint32_t intrl2_1_cpu_stat; // 0x240 + uint32_t intrl2_1_cpu_set; // 0x244 + uint32_t intrl2_1_cpu_clear; // 0x248 + uint32_t intrl2_1_cpu_stat_mask; // 0x24C + uint32_t intrl2_1_cpu_set_mask; // 0x250 + uint32_t intrl2_1_cpu_clear_mask; // 0x254 + uint8_t intrl2_unused2[168]; // 0x258-0x300 + uint32_t rbuf_ctrl; // 0x300 + uint8_t rbuf_unused1[176]; // 0x304-0x3B4 + uint32_t rbuf_tbuf_size_ctrl; // 0x3B4 + uint8_t rbuf_unused2[584]; // 0x3B8-0x600 + uint32_t tbuf_ctrl; // 0x600 + uint32_t tbuf_ck_ctrl; // 0x604 + uint8_t tbuf_unused[504]; // 0x608-0x800 + uint8_t umac_unused1[8]; // 0x800-0x808 + uint32_t umac_cmd; // 0x808 + uint32_t umac_mac0; // 0x80C + uint32_t umac_mac1; // 0x810 + uint32_t umac_max_frame_len; // 0x814 + uint8_t umac_unused2[796]; // 0x818-0xB34 + uint32_t umac_tx_flush; // 0xB34 + uint8_t umac_unused3[584]; // 0xB38-0xD80 + uint32_t umac_mib_ctrl; // 0xD80 + uint8_t umac_unused4[144]; // 0xD84-0xE14 + uint32_t umac_mdio_cmd; // 0xE14 + uint8_t unused1[56]; // 0xE18-0xE50 + uint32_t umac_mdf_ctrl; // 0xE50 + uint8_t unused2[4524]; // 0xE54-0x2000 + struct genet_dma dma_rx; // 0x2000-0x30B0 + uint8_t unused3[3920]; // 0x30B0-0x4000 + struct genet_dma dma_tx; // 0x4000 +}; + +#define MBOX_REQUEST 0 +#define MBOX_TAG_HARDWARE_GET_MAC_ADDRESS 0x00010003 +#define MBOX_TAG_HARDWARE_GET_CLK_RATE 0x00030002 +#define MBOX_TAG_HARDWARE_SET_CLK_RATE 0x00038002 +#define MBOX_TAG_LAST 0 +#define MBOX_ADDR 0x08000000 +#define MBOX_RESPONSE 0x80000000 +#define MBOX_FULL 0x80000000 +#define MBOX_EMPTY 0x40000000 + +struct mbox_regs { + uint32_t read; // 0x00 + uint32_t unused1[3]; // 0x04-0x10 + uint32_t poll; // 0x10 + uint32_t sender; // 0x14 + uint32_t status; // 0x18 + uint32_t config; // 0x1C + uint32_t write; // 0x20 +}; + +/* TSB structure for BCM GENET hardware checksum offload */ +struct bcmgenet_tsb { + uint32_t length_status; /* length and peripheral status */ + uint32_t ext_status; /* Extended status*/ + uint32_t rx_csum; /* partial rx checksum */ + uint32_t unused1[9]; /* unused */ + /** + * Tx checksum info. + * Bits 14-0 contain the checksum destination offset. + * Bit 15 calculate UDP/TCP checksum. + * Bits 30-16 contain the offset of first byte to be checksummed. + * Bit 31 enables checksum offloading. + * Note the transport layer header must be initialised with pseudo header checksum. + */ + uint32_t tx_csum_info; + uint32_t unused2[3]; /* unused */ +}; + +#define CHECKSUM_TSB_LENGTH sizeof(struct bcmgenet_tsb) diff --git a/examples/echo_server/echo.c b/examples/echo_server/echo.c index 483a795e..5a232f08 100644 --- a/examples/echo_server/echo.c +++ b/examples/echo_server/echo.c @@ -19,6 +19,7 @@ #include "lwip/pbuf.h" #include "echo.h" +#include "pseudo_checksum.h" __attribute__((__section__(".serial_client_config"))) serial_client_config_t serial_config; @@ -67,9 +68,13 @@ void init(void) net_queue_init(&net_tx_handle, net_config.tx.free_queue.vaddr, net_config.tx.active_queue.vaddr, net_config.tx.num_buffers); net_buffers_init(&net_tx_handle, 0); - +#if defined(CONFIG_PLAT_BCM2711) + sddf_lwip_init(&lib_sddf_lwip_config, &net_config, &timer_config, net_rx_handle, net_tx_handle, NULL, NULL, + netif_status_callback, NULL, pbuf_needs_checksum, add_checksum_and_transmit); +#else sddf_lwip_init(&lib_sddf_lwip_config, &net_config, &timer_config, net_rx_handle, net_tx_handle, NULL, NULL, netif_status_callback, NULL, NULL, NULL); +#endif set_timeout(); setup_udp_socket(); diff --git a/examples/echo_server/echo.mk b/examples/echo_server/echo.mk index a848f55c..7358e74a 100644 --- a/examples/echo_server/echo.mk +++ b/examples/echo_server/echo.mk @@ -23,7 +23,8 @@ SUPPORTED_BOARDS := \ qemu_virt_riscv64 \ rock3b \ star64 \ - x86_64_generic + x86_64_generic \ + rpi4b_1gb TOOLCHAIN ?= clang MICROKIT_CONFIG ?= debug @@ -73,7 +74,8 @@ LIBS := --start-group -lmicrokit -Tmicrokit.ld libsddf_util_debug.a \ --end-group ECHO_OBJS := echo.o utilization_socket.o \ - udp_echo_socket.o tcp_echo_socket.o + udp_echo_socket.o tcp_echo_socket.o \ + pseudo_checksum.o DEPS := $(ECHO_OBJS:.o=.d) diff --git a/examples/echo_server/include/lwip/lwipopts.h b/examples/echo_server/include/lwip/lwipopts.h index 1254b461..fa6e866f 100644 --- a/examples/echo_server/include/lwip/lwipopts.h +++ b/examples/echo_server/include/lwip/lwipopts.h @@ -102,6 +102,15 @@ #define CHECKSUM_GEN_ICMP 0 #define CHECKSUM_GEN_ICMP6 0 +#elif defined(NETWORK_HW_HAS_TRANSPORT_CHECKSUM) + +/* Hw generates transport layer checksums only */ +#define CHECKSUM_GEN_IP 1 +#define CHECKSUM_GEN_UDP 0 +#define CHECKSUM_GEN_TCP 0 +#define CHECKSUM_GEN_ICMP 0 +#define CHECKSUM_GEN_ICMP6 0 + #else #define CHECKSUM_GEN_IP 1 diff --git a/examples/echo_server/include/pseudo_checksum.h b/examples/echo_server/include/pseudo_checksum.h new file mode 100644 index 00000000..bc680af0 --- /dev/null +++ b/examples/echo_server/include/pseudo_checksum.h @@ -0,0 +1,49 @@ +/* + * Copyright 2022, UNSW + * SPDX-License-Identifier: BSD-2-Clause + */ + +#pragma once + +#include +#include +#include "lwip/pbuf.h" + +/** + * This library supports hardware checksum generation for the GENET NIC. For + * GENET NIC checksum offload, the first bytes before each packet must contain + * metadata pertaining to where the checksum should be written within the + * packet, as well as which bytes the checksum should be summed over. + * + * Additionally, for transport layer protocols requiring a pseudo-header + * checksum, this must already be present in the transport layer checksum + * header. + * + * The library works by appending an lwip pbuf containing this metadata to each + * outgoing packet, and, if needed, setting the transport layer checksum to the + * pseudo-header checksum. + */ + + /** + * Lib sDDF lwIP function for `tx_intercept_condition`. Checks whether the + * checksum metadata pbuf has been appended to the outgoing pbuf. + * + * @param p pbuf to be transmitted via sDDF. + * + * @return true checksum has NOT been appended, thus transmission must be + * intercepted. + * @return false checksum has been appended, transmission may continue. + */ +bool pbuf_needs_checksum(struct pbuf *p); + +/** + * Lib sDDF lwIP function for `tx_handle_intercept`. Appends a static pbuf + * containing checksum metadata to the provided outgoing pbuf p, then transmits + * the pbuf using `sddf_lwip_transmit_pbuf`. If the outgoing pbuf contains a + * packet which requires a pseudo-header calculation, this function will add the + * partial checksum to the transport-layer header. + * + * @param p pbuf to be transmitted via sDDF. + * @return net_sddf_err_t error result of concatenated pbuf transmission. + */ +net_sddf_err_t add_checksum_and_transmit(struct pbuf *p); diff --git a/examples/echo_server/meta.py b/examples/echo_server/meta.py index cf210eb6..9cd0e3ad 100644 --- a/examples/echo_server/meta.py +++ b/examples/echo_server/meta.py @@ -195,7 +195,7 @@ def generate( assert timer_node is not None timer_driver = ProtectionDomain( - "timer_driver", "timer_driver.elf", priority=101, cpu=get_core("timer_driver") + "timer_driver", "timer_driver.elf", priority=102, cpu=get_core("timer_driver") ) timer_system = Sddf.Timer(sdf, timer_node, timer_driver) @@ -273,6 +273,13 @@ def generate( ethernet_driver.add_map( Map(clock_controller, 0x3000000, perms="rw", cached=False) ) + elif board.name == "rpi4b_1gb": + # Ethernet driver requires timer access to wait for reconfiguration + timer_system.add_client(ethernet_driver) + + mbox = MemoryRegion(sdf, "mbox", 0x10_000, paddr=0xFE00B000) + sdf.add_mr(mbox) + ethernet_driver.add_map(Map(mbox, 0x3000000, perms="rw", cached=False)) if board.arch == SystemDescription.Arch.X86_64: hw_net_rings = SystemDescription.MemoryRegion( @@ -483,6 +490,11 @@ def generate( assert client1_lib_sddf_lwip.connect() assert client1_lib_sddf_lwip.serialise_config(output_dir) + if board.name == "rpi4b_1gb": + update_elf_section( + "eth_driver.elf", "timer_client_config", "timer_client_ethernet_driver" + ) + with open(f"{output_dir}/benchmark_client_config.data", "wb+") as f: f.write(bench_client_config.serialise()) update_elf_section( diff --git a/examples/echo_server/pseudo_checksum.c b/examples/echo_server/pseudo_checksum.c new file mode 100644 index 00000000..0778dc36 --- /dev/null +++ b/examples/echo_server/pseudo_checksum.c @@ -0,0 +1,154 @@ +/* + * Copyright 2022, UNSW + * SPDX-License-Identifier: BSD-2-Clause + */ + +#include +#include +#include +#include +#include +#include +#include "pseudo_checksum.h" +#include "lwip/pbuf.h" + +/* Shared ethernet.h definitions from GENET driver */ + +#define STATUS_TX_CSUM_PROTO_UDP BIT(15) /* Calculate UDP/TCP checksum */ +#define STATUS_TX_CSUM_LV BIT(31) /* Enable transmit checksum offloading */ + +/* TSB structure for BCM GENET hardware checksum offload */ +struct bcmgenet_tsb { + uint32_t length_status; /* length and peripheral status */ + uint32_t ext_status; /* Extended status*/ + uint32_t rx_csum; /* partial rx checksum */ + uint32_t unused1[9]; /* unused */ + /** + * Tx checksum info. + * Bits 14-0 contain the checksum destination offset. + * Bit 15 calculate UDP/TCP checksum. + * Bits 30-16 contain the offset of first byte to be checksummed. + * Bit 31 enables checksum offloading. + * Note the transport layer header must be initialised with pseudo header checksum. + */ + uint32_t tx_csum_info; + uint32_t unused2[3]; /* unused */ +}; + +struct bcmgenet_tsb device_checksum; + +/* Temporary pbuf used to pre-pend packets with TSB data. */ +struct pbuf checksum_pbuf; + +/* Pseudo-header used for UDP/TCP checksum calculation */ +typedef struct pseudo_header { + uint32_t src_ip; + uint32_t dst_ip; + /* Always set to 0 */ + uint8_t reserved; + uint8_t protocol; + /* Transport layer packet length */ + uint16_t len; +} pseudo_header_t; + +/* Minimum ethernet frame size */ +#define MIN_ETH_PKT_LEN 60 + +bool pbuf_needs_checksum(struct pbuf *p) +{ + if (p == &checksum_pbuf) { + return false; + } + return true; +} + +net_sddf_err_t add_checksum_and_transmit(struct pbuf *p) +{ + /* Construct pbuf to prepend to packet pbuf */ + checksum_pbuf.payload = &device_checksum; + checksum_pbuf.len = sizeof(struct bcmgenet_tsb); + checksum_pbuf.next = p; + checksum_pbuf.tot_len = sizeof(struct bcmgenet_tsb) + p->tot_len; + + struct ethernet_header *eth_hdr = (struct ethernet_header *)p->payload; + switch (eth_hdr->type) { + case HTONS(ETH_TYPE_ARP): { + /** + * No checksum is needed for ARP packets, so we point the hardware to + * the end of the packet to start calculation. If the packet is less + * than the minimum 60-byte frame, we point to the minimum padded size. + * */ + device_checksum.tx_csum_info = MAX(MIN_ETH_PKT_LEN, p->tot_len) << 16 | MAX(MIN_ETH_PKT_LEN, p->tot_len); + break; + } + case HTONS(ETH_TYPE_IP): { + struct ipv4_header *ip_hdr = (struct ipv4_header *)((void *)p->payload + sizeof(struct ethernet_header)); + + bool supported = true; + bool pseudo = true; + uintptr_t checksum_addr = 0; + switch (ip_hdr->protocol) { + case IP_PROTOCOL_UDP: { + struct udp_header *udp_hdr = (struct udp_header *)ipv4_payload_start(ip_hdr); + /* UDP header checksum must lie within this pbuf's payload so pseudo header checksum can be written */ + assert(p->len >= sizeof(struct ethernet_header) + ipv4_header_length(ip_hdr) + sizeof(struct udp_header)); + checksum_addr = (uintptr_t)&udp_hdr->check; + break; + } + case IP_PROTOCOL_TCP: { + struct tcp_header *tcp_hdr = (struct tcp_header *)ipv4_payload_start(ip_hdr); + /* TCP header checksum must lie within this pbuf's payload so pseudo header checksum can be written */ + assert(p->len >= sizeof(struct ethernet_header) + ipv4_header_length(ip_hdr) + sizeof(struct tcp_header)); + checksum_addr = (uintptr_t)&tcp_hdr->check; + break; + } + case IP_PROTOCOL_ICMP: { + struct icmp_header *icmp_hdr = (struct icmp_header *)ipv4_payload_start(ip_hdr); + checksum_addr = (uintptr_t)&icmp_hdr->check; + pseudo = false; + break; + } + default: + supported = false; + break; + } + + if (!supported) { + device_checksum.tx_csum_info = MAX(MIN_ETH_PKT_LEN, p->tot_len) << 16 | MAX(MIN_ETH_PKT_LEN, p->tot_len); + break; + } + + if (pseudo) { + /* Construct pseudo header */ + pseudo_header_t pseudo_data = { ip_hdr->src_ip, ip_hdr->dst_ip, 0, ip_hdr->protocol, + HTONS(ipv4_payload_length(ip_hdr)) }; + + /* Sum up the pseudo-header */ + uint32_t sum = 0; + uint16_t *pseudo_data_ptr = (uint16_t *)&pseudo_data; + for (uint8_t i = 0; i < sizeof(pseudo_header_t) / sizeof(uint16_t); i++) { + sum += pseudo_data_ptr[i]; + } + + /* Fold 32-bit sum to 16 bits (one's complement sum) */ + while (sum >> 16) { + sum = (sum & 0xFFFF) + (sum >> 16); + } + + /* Set packet checksum to pseudo checksum */ + *(uint16_t *)checksum_addr = sum; + } + + device_checksum.tx_csum_info = STATUS_TX_CSUM_LV | STATUS_TX_CSUM_PROTO_UDP + | checksum_addr - (uintptr_t)p->payload // checksum destination offset + | ((uintptr_t)ipv4_payload_start(ip_hdr) - (uintptr_t)p->payload) + << 16; // start byte + break; + } + default: + device_checksum.tx_csum_info = MAX(MIN_ETH_PKT_LEN, p->tot_len) << 16 | MAX(MIN_ETH_PKT_LEN, p->tot_len); + break; + } + + return sddf_lwip_transmit_pbuf(&checksum_pbuf); +} diff --git a/include/sddf/network/constants.h b/include/sddf/network/constants.h index 1edb36c6..150880b0 100644 --- a/include/sddf/network/constants.h +++ b/include/sddf/network/constants.h @@ -7,12 +7,16 @@ #include #include +#include #define ETH_TYPE_ARP 0x0806U #define ETH_TYPE_IP 0x0800U #define ETH_HWADDR_LEN 6 #define ETHARP_OPCODE_REQUEST 1 #define ETHARP_OPCODE_REPLY 2 +#define IP_PROTOCOL_ICMP 0x01 +#define IP_PROTOCOL_TCP 0x06 +#define IP_PROTOCOL_UDP 0x11 #define NET_BUFFER_SIZE 2048 @@ -26,6 +30,94 @@ struct ethernet_header { uint16_t type; } __attribute__((packed)); +struct ipv4_header { + uint8_t ihl : 4; + uint8_t version : 4; + uint8_t ecn : 2; + uint8_t dscp : 6; + uint16_t tot_len; + uint16_t id; + uint8_t frag_offset1 : 5; + uint8_t more_frag : 1; + uint8_t no_frag : 1; + uint8_t reserved : 1; + uint8_t frag_offset2; + uint8_t ttl; + uint8_t protocol; + uint16_t check; + uint32_t src_ip; + uint32_t dst_ip; +} __attribute__((__packed__)); + +struct udp_header { + uint16_t src_port; + uint16_t dst_port; + uint16_t len; + uint16_t check; +} __attribute__((__packed__)); + +struct tcp_header { + uint16_t src_port; + uint16_t dst_port; + uint32_t seq; + uint32_t ack_seq; + uint16_t reserved : 4; + uint16_t doff : 4; + uint16_t fin : 1; + uint16_t syn : 1; + uint16_t rst : 1; + uint16_t psh : 1; + uint16_t ack : 1; + uint16_t urg : 1; + uint16_t ece : 1; + uint16_t cwr : 1; + uint16_t window; + uint16_t check; + uint16_t urg_ptr; +} __attribute__((__packed__)); + +struct icmp_header { + uint8_t type; + uint8_t code; + uint16_t check; +} __attribute__((__packed__)); + +/** + * IPv4 header length in bytes. + * + * @param ip_hdr address of IPv4 header. + * + * @return IPv4 header length in bytes. + */ +static inline uint8_t ipv4_header_length(struct ipv4_header *ip_hdr) +{ + return 4 * ip_hdr->ihl; +} + +/** + * IPv4 payload length in bytes. + * + * @param ip_hdr address of IPv4 header. + * + * @return IPv4 payload length in bytes. + */ +static inline uint16_t ipv4_payload_length(struct ipv4_header *ip_hdr) +{ + return HTONS(ip_hdr->tot_len) - ipv4_header_length(ip_hdr); +} + +/** + * Extract start address of IPv4 payload. + * + * @param ip_hdr address of IPv4 packet. + * + * @return address of IPv4 payload. + */ +static inline void *ipv4_payload_start(struct ipv4_header *ip_hdr) +{ + return (uint8_t *)ip_hdr + ipv4_header_length(ip_hdr); +} + /* * By default we assume that the hardware we are dealing with * cannot generate checksums on transmit. We use this macro @@ -35,4 +127,13 @@ struct ethernet_header { || defined(CONFIG_PLAT_ODROIDC4) || defined(CONFIG_PLAT_STAR64) || defined(CONFIG_PLAT_HIFIVE_P550) \ || defined(CONFIG_PLAT_ODROIDC2) #define NETWORK_HW_HAS_CHECKSUM +#elif defined(CONFIG_PLAT_BCM2711) +#define NETWORK_HW_HAS_TRANSPORT_CHECKSUM +#endif + +#if defined(CONFIG_PLAT_BCM2711) +#define PBUF_LINK_ENCAPSULATION_HLEN 64 +#define NETWORK_HW_HAS_TRANSPORT_CHECKSUM +#else +#define PBUF_LINK_ENCAPSULATION_HLEN 0 #endif diff --git a/tools/make/board/rpi4b_1gb.mk b/tools/make/board/rpi4b_1gb.mk index a37bdfd5..36ab4efb 100644 --- a/tools/make/board/rpi4b_1gb.mk +++ b/tools/make/board/rpi4b_1gb.mk @@ -5,8 +5,8 @@ # # Set up variables for Raspberry Pi 4b ith 1Gb RAM # Should be included _before_ toolchain makefile. -# NET_DRIV_DIR := -# ETH_DRIV := eth_driver_dwmac-5.10a.elf +NET_DRIV_DIR := genet +ETH_DRIV := eth_driver_genet.elf UART_DRIV_DIR := ns16550a TIMER_DRIV_DIR := bcm2835 #I2C_DRIV_DIR := ${PLATFORM} diff --git a/tools/meta/board.py b/tools/meta/board.py index 809ceee2..e96a0e7b 100644 --- a/tools/meta/board.py +++ b/tools/meta/board.py @@ -137,6 +137,7 @@ class Board: paddr_top=0x2_000_000, serial="soc/serial@7e215040", timer="soc/timer@7e003000", + ethernet="scb/ethernet@7d580000", ), Board( name="serengeti",