From 84b81a959c1d230160d9eb2fe8bac40dbfdd050d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jean-Fran=C3=A7ois=20Brisson?= <281253927+sparkainlp-x@users.noreply.github.com> Date: Sun, 27 Sep 2026 11:38:41 -0400 Subject: [PATCH] docs: English-only docs, fix run_hls.tcl line endings, actions v7, CHANGELOG --- .github/workflows/ci.yml | 10 ++--- CHANGELOG.md | 28 +++++++++++++ CMakeLists.txt | 8 ++-- Makefile | 32 +++++++-------- README.md | 68 ++++++++++++++++--------------- docs/deployment-zcu111.md | 42 ++++++++++--------- drivers/axi_dma_driver.cpp | 10 ++--- hil/hil_benchmark.cpp | 10 ++--- hls/run_hls.tcl | 29 ++++++++++++- hls/run_hls_400mhz.tcl | 2 +- main.cpp | 4 +- vivado/README.md | 18 ++++---- vivado/create_bd.tcl | 14 +++---- vivado/optimize_timing_400mhz.tcl | 10 ++--- 14 files changed, 173 insertions(+), 112 deletions(-) create mode 100644 CHANGELOG.md diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 33cc3e9..72dd4c8 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -22,7 +22,7 @@ jobs: steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@v7 - name: Install build dependencies run: | @@ -50,7 +50,7 @@ jobs: run: ./build/ldpc_benchmark hardware-bitstream-build: - # Ne jamais exécuter du code d’une pull request externe sur un runner local. + # Never run code from an external pull request on a local runner. # The trusted hardware runner is intentionally opt-in to avoid queued CI when # a ZCU111/Vivado runner is offline. Start it manually with run_hardware=true. if: github.event_name == 'workflow_dispatch' && inputs.run_hardware @@ -60,7 +60,7 @@ jobs: steps: - name: Checkout repository - uses: actions/checkout@v4 + uses: actions/checkout@v7 - name: Build HLS, Vivado and PetaLinux artifacts shell: bash @@ -87,7 +87,7 @@ jobs: run: ./build-hil/hil_benchmark - name: Upload HIL latency report - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v7 with: name: qldpc-hil-latency-report if-no-files-found: error @@ -95,7 +95,7 @@ jobs: path: latencies_report.csv - name: Upload FPGA and PetaLinux artifacts - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v7 with: name: qldpc-rfsoc-bitstream-artifacts if-no-files-found: error diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..4d6884d --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,28 @@ +# Changelog + +All notable changes to this project are documented here. The format follows +[Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and the project uses +[Semantic Versioning](https://semver.org/). + +## [Unreleased] + +### Fixed +- `hls/run_hls.tcl` was stored as a single line with literal `\n` sequences, so Tcl read the whole file as one comment and the script did nothing. It is now a normal multi-line script (synthesis is still UNRUN in CI). + +### Changed +- Documentation is English throughout: the README's French section, `docs/deployment-zcu111.md`, `vivado/README.md`, Makefile messages, and code comments and console strings. Hardware documents carry explicit UNRUN / TARGET status notes. +- CI actions bumped to `actions/checkout@v7` and `actions/upload-artifact@v7` (Node 24). + +## [0.1.1] - 2026-09-26 + +### Changed +- `CITATION.cff` version bump for the Zenodo archival release (DOI 10.5281/zenodo.22985528; concept DOI 10.5281/zenodo.22985527). + +## [0.1.0] - 2026-09-26 + +### Added +- English-first README, evidence tags (the ~70 ns GF(2) micro-benchmark is REPORTED; host unspecified), MIT license, `CITATION.cff`. +- CMake + Catch2 software CI with an opt-in self-hosted hardware job. + +### Removed +- Promotional drafts; `verify_correction()` limitation documented (FER is 0 by construction until replaced). diff --git a/CMakeLists.txt b/CMakeLists.txt index 0f9c52b..8415a5a 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,22 +1,22 @@ cmake_minimum_required(VERSION 3.20) project(qldpc_decoder_cpp LANGUAGES CXX) -# La version actuelle de ldpc utilise des fonctionnalités C++20. +# The current ldpc release uses C++20 features. set(CMAKE_CXX_STANDARD 20) set(CMAKE_CXX_STANDARD_REQUIRED ON) set(CMAKE_CXX_EXTENSIONS OFF) -# Optimisations adaptées à une compilation Release sur la machine hôte. +# Optimisations for a Release build on the host machine. if(MSVC) add_compile_options(/O2 /arch:AVX2) else() add_compile_options(-O3 -march=native -ffast-math -flto) endif() -# OpenMP est requis pour les chemins parallélisés de l’application. +# OpenMP is required for the parallel code paths. find_package(OpenMP REQUIRED) -# Bibliothèque C++ ldpc de Joschka Roffe. +# Joschka Roffe's ldpc C++ library. include(FetchContent) FetchContent_Declare( ldpc diff --git a/Makefile b/Makefile index 6c7101b..303765e 100644 --- a/Makefile +++ b/Makefile @@ -1,5 +1,5 @@ # ================================================================================ -# Pipeline matériel et embarqué qLDPC : HLS -> Vivado -> PetaLinux +# qLDPC hardware and embedded pipeline: HLS -> Vivado -> PetaLinux (UNRUN) # ================================================================================ HLS_DIR := hls @@ -18,33 +18,33 @@ help: @echo "====================================================================" @echo " Makefile qLDPC - Pipeline ZCU111 RFSoC" @echo "====================================================================" - @echo " make hls Synthèse Vitis HLS à 300 MHz" - @echo " make hls-400mhz Synthèse Vitis HLS à 400 MHz" - @echo " make vivado Génération du Block Design et export XSA" - @echo " make petalinux Compilation de l’image PetaLinux" - @echo " make all Exécution complète HLS -> Vivado -> PetaLinux" - @echo " make clean Suppression des artefacts matériels" + @echo " make hls Vitis HLS synthesis at 300 MHz (TARGET)" + @echo " make hls-400mhz Vitis HLS synthesis at 400 MHz (TARGET)" + @echo " make vivado Generate Block Design and export XSA" + @echo " make petalinux Build the PetaLinux image" + @echo " make all Full run HLS -> Vivado -> PetaLinux" + @echo " make clean Remove hardware build artifacts" @echo "====================================================================" all: hls vivado petalinux - @echo ">>> SUCCESS : pipeline complet exécuté." + @echo ">>> Pipeline finished." hls: - @echo ">>> [1/3] Synthèse Vitis HLS 300 MHz..." + @echo ">>> [1/3] Vitis HLS synthesis, 300 MHz..." $(VITIS_HLS) -f $(HLS_DIR)/run_hls.tcl hls-400mhz: - @echo ">>> Synthèse Vitis HLS 400 MHz..." + @echo ">>> Vitis HLS synthesis, 400 MHz..." $(VITIS_HLS) -f $(HLS_DIR)/run_hls_400mhz.tcl vivado: hls - @echo ">>> [2/3] Génération Vivado du Block Design..." + @echo ">>> [2/3] Generating Vivado Block Design..." cd $(VIVADO_DIR) && $(VIVADO) -mode batch -source create_bd.tcl petalinux: vivado - @echo ">>> [3/3] Compilation PetaLinux..." + @echo ">>> [3/3] Building PetaLinux..." @if [ ! -d "$(PETALINUX_DIR)" ]; then \ - echo "Erreur: $(PETALINUX_DIR) est absent; initialisez d’abord un projet PetaLinux."; \ + echo "Error: $(PETALINUX_DIR) is missing; initialise a PetaLinux project first."; \ exit 1; \ fi cd $(PETALINUX_DIR) && \ @@ -54,12 +54,12 @@ petalinux: vivado --fsbl images/linux/zynqmp_fsbl.elf \ --u-boot images/linux/u-boot.elf \ --fpga images/linux/download.bit --force - @echo ">>> Artifact final : $(PETALINUX_DIR)/images/linux/BOOT.BIN" + @echo ">>> Final artifact: $(PETALINUX_DIR)/images/linux/BOOT.BIN" clean: - @echo ">>> Nettoyage des artefacts matériels..." + @echo ">>> Removing hardware build artifacts..." rm -rf $(HLS_DIR)/qldpc_hls_project $(HLS_DIR)/qldpc_hls_project_400mhz rm -rf $(HLS_DIR)/*.log $(HLS_DIR)/vivado* $(HLS_DIR)/.Xil rm -rf $(VIVADO_DIR)/qldpc_vivado_bd $(VIVADO_DIR)/*.jou $(VIVADO_DIR)/*.log rm -rf $(VIVADO_DIR)/*.xsa $(VIVADO_DIR)/.Xil - @echo ">>> Nettoyage terminé." + @echo ">>> Clean finished." diff --git a/README.md b/README.md index 8a59a4a..06fd7fa 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ C++/HLS **research scaffold** for a qLDPC decoder: CMake + Catch2 + a sparse GF( [![Hardware results: UNRUN](https://img.shields.io/badge/hardware%20results-UNRUN-lightgrey.svg)](#evidence-tags) [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.22985527.svg)](https://doi.org/10.5281/zenodo.22985527) -*English first; the original French documentation follows below. / La documentation originale en français suit.* +*English documentation. A French translation can be provided on request.* ## What it is @@ -39,8 +39,8 @@ cmake --build build --parallel Expected output: ```text -qldpc_decoder_cpp: bibliothèque ldpc chargée avec succès. -OpenMP est activé au niveau de la cible CMake. +qldpc_decoder_cpp: ldpc library loaded. +OpenMP is enabled on the CMake target. ``` `CMakeLists.txt` uses `-O3 -march=native -ffast-math -flto` on non-MSVC compilers, so the binary is tuned to the build machine. @@ -87,79 +87,81 @@ Citation metadata is in [CITATION.cff](CITATION.cff); GitHub shows a "Cite this --- -## Description en français +## Build, benchmark and hardware flow (details) -Exemple C++ minimal utilisant la bibliothèque [`ldpc`](https://github.com/quantumgizmos/ldpc) de Joschka Roffe via CMake `FetchContent`. La configuration active OpenMP ainsi que les optimisations processeur en mode non-MSVC. +Minimal C++ example that uses Joschka Roffe's [`ldpc`](https://github.com/quantumgizmos/ldpc) library through CMake `FetchContent`. OpenMP and CPU optimisations are enabled for non-MSVC compilers. -### Prérequis +### Prerequisites -Il faut disposer de CMake 3.20 ou ultérieur, d’un compilateur C++ compatible C++20, de Git et d’une implémentation OpenMP. +CMake 3.20 or later, a C++20 compiler, Git, and an OpenMP implementation. -### Compilation +### Build ```bash cmake -S . -B build -DCMAKE_BUILD_TYPE=Release cmake --build build --parallel ``` -### Exécution +### Run ```bash ./build/qldpc_decoder ``` -### Benchmark de latence +### Latency micro-benchmark -Le projet construit également `ldpc_benchmark`, qui mesure une multiplication matrice-vecteur sparse sur GF(2) après une phase d’échauffement. Il exécute 31 échantillons de 1 000 itérations, puis affiche la latence médiane et le 95e percentile en nanosecondes. Les valeurs d'environ 70 ns observées jusqu'ici sont **REPORTED ; hôte/conditions non précisés** et concernent une micro-opération, pas le décodage qLDPC complet. +The project also builds `ldpc_benchmark`, which times a sparse GF(2) matrix-vector product after a warm-up phase. It runs 31 samples of 1,000 iterations and prints the median and 95th-percentile latency in nanoseconds. The values of about 70 ns observed so far are **REPORTED; host/conditions unspecified** and concern a single micro-operation, not full qLDPC decoding. ```bash ./build/ldpc_benchmark ``` -Dans GitHub Actions, le benchmark échoue si la médiane atteint ou dépasse `1000 ns` (`1 µs`). Le seuil peut être changé avec la variable d’environnement `LDPC_MAX_LATENCY_NS`. Les runners GitHub-hosted étant virtualisés et partagés, le résultat est un contrôle de régression indicatif et ne remplace pas une mesure sur machine dédiée. +In GitHub Actions the benchmark fails if the median reaches or exceeds `1000 ns` (`1 µs`). The threshold can be changed with the `LDPC_MAX_LATENCY_NS` environment variable. GitHub-hosted runners are virtualised and shared, so this is an indicative regression check, not a measurement on dedicated hardware. -La première configuration télécharge automatiquement le dépôt amont `quantumgizmos/ldpc` dans le répertoire de build. Le fichier `CMakeLists.txt` utilise `-O3 -march=native -ffast-math -flto` sur les compilateurs non-MSVC et `/O2 /arch:AVX2` sous MSVC. +The first configure step downloads the upstream `quantumgizmos/ldpc` repository into the build directory. `CMakeLists.txt` uses `-O3 -march=native -ffast-math -flto` on non-MSVC compilers and `/O2 /arch:AVX2` on MSVC. -> `-march=native` produit un binaire adapté à la machine de compilation. Pour distribuer le binaire sur d’autres processeurs, remplacez ce drapeau par une architecture cible portable. +> `-march=native` produces a binary tuned to the build machine. To distribute the binary to other CPUs, replace this flag with a portable target architecture. -### Synthèse AMD Vitis HLS +### AMD Vitis HLS synthesis (UNRUN) -Le répertoire `hls/` contient le noyau `qldpc_decode_kernel`, son testbench et le script `run_hls.tcl` ciblant le RFSoC ZCU111 (`xczu28dr-ffvg1517-2-e`) à 300 MHz. Depuis un environnement où Vitis HLS 2023.2 est installé : +The `hls/` directory contains the `qldpc_decode_kernel` kernel, its testbench and the `run_hls.tcl` script targeting the RFSoC ZCU111 (`xczu28dr-ffvg1517-2-e`) at 300 MHz (**TARGET**). From an environment with Vitis HLS 2023.2 installed: ```bash source /tools/Xilinx/Vitis_HLS/2023.2/settings64.sh vitis_hls -f hls/run_hls.tcl ``` -La synthèse FPGA n’est pas exécutée par la CI GitHub Actions standard, car le runner ne fournit ni Vitis HLS ni les bibliothèques AMD HLS. La CI vérifie en revanche les sources C++ portables, les tests Catch2 et le benchmark logiciel. +FPGA synthesis is not run by the standard GitHub Actions CI, because the hosted runner provides neither Vitis HLS nor the AMD HLS libraries. CI does verify the portable C++ sources, the Catch2 tests and the software micro-benchmark. -### Pipeline matériel avec Make +### Hardware pipeline with Make (UNRUN) -Le [`Makefile`](./Makefile) orchestre les étapes matérielles lorsqu’un environnement AMD est installé : +The [`Makefile`](./Makefile) orchestrates the hardware steps when an AMD toolchain is installed: ```bash make help -make hls # solution HLS 300 MHz -make hls-400mhz # solution HLS 400 MHz -make vivado # Block Design et export XSA -make petalinux # image Linux et BOOT.BIN -make all # chaîne complète +make hls # 300 MHz HLS solution +make hls-400mhz # 400 MHz HLS solution +make vivado # Block Design and XSA export +make petalinux # Linux image and BOOT.BIN +make all # full chain ``` -Avant son utilisation, charger les environnements Vitis/Vivado et PetaLinux correspondant à l’installation locale. Le dépôt fournit uniquement les sources et scripts ; il ne contient pas un projet PetaLinux initialisé ni les outils AMD. +Source the Vitis/Vivado and PetaLinux environments for your local installation first. The repository provides sources and scripts only; it does not contain an initialised PetaLinux project or the AMD tools. -### CI matérielle sur runner auto-hébergé +### Hardware CI on a self-hosted runner (opt-in) -Le workflow contient un job `hardware-bitstream-build` qui s’exécute uniquement sur un runner GitHub auto-hébergé portant les labels `self-hosted`, `vivado` et `zcu111`. Ce runner doit disposer de Vitis, Vitis HLS, Vivado, PetaLinux, des licences AMD nécessaires et d’un projet PetaLinux initialisé dans `petalinux/`. +The workflow contains a `hardware-bitstream-build` job that runs only on a self-hosted GitHub runner labelled `self-hosted`, `vivado` and `zcu111`. That runner needs Vitis, Vitis HLS, Vivado, PetaLinux, the required AMD licences, and an initialised PetaLinux project in `petalinux/`. -Le job matériel attend la réussite de `software-ci`, lance `make all`, puis publie les fichiers `.xsa`, `BOOT.BIN`, `image.ub` et `download.bit` comme artefacts GitHub Actions. Pour protéger la machine locale, il n’est pas déclenché par les pull requests : les changements doivent d’abord être fusionnés dans `main`, ou le workflow doit être lancé manuellement par un opérateur de confiance. +The hardware job waits for `software-ci` to pass, runs `make all`, and publishes the `.xsa`, `BOOT.BIN`, `image.ub` and `download.bit` files as GitHub Actions artifacts. To protect the local machine it is not triggered by pull requests: changes must first be merged into `main`, or a trusted operator must start the workflow manually. -### Test HIL automatisé (UNRUN) +### Automated HIL test (UNRUN) -> **Statut : UNRUN.** Aucun test HIL n'a été exécuté ni publié. `verify_correction()` retourne toujours `true` (voir « Known limitations » ci-dessus) : tant qu'elle n'est pas remplacée, le FER rapporté vaut 0 par construction. +> **Status: UNRUN.** No HIL test has been run or published. `verify_correction()` always returns `true` (see "Known limitations" above); until it is replaced, the reported FER is 0 by construction. -Le benchmark [`hil/hil_benchmark.cpp`](./hil/hil_benchmark.cpp) exécute 100 000 transferts AXI-DMA/FPGA, mesure chaque aller-retour en nanosecondes, exporte `latencies_report.csv` et vérifie la latence maximale. +The benchmark [`hil/hil_benchmark.cpp`](./hil/hil_benchmark.cpp) performs 100,000 AXI-DMA/FPGA transfers, times each round trip in nanoseconds, exports `latencies_report.csv` and checks the maximum latency. -Le job `hardware-bitstream-build` de GitHub Actions compile ce binaire avec `BUILD_HIL_BENCHMARK=ON`, puis l’exécute avec `HIL_MAX_LATENCY_US=100`. Le job échoue dès qu’une mesure dépasse 100 µs et conserve le CSV comme artefact `qldpc-hil-latency-report`. +The `hardware-bitstream-build` job compiles this binary with `BUILD_HIL_BENCHMARK=ON` and runs it with `HIL_MAX_LATENCY_US=100`. The job fails as soon as one measurement exceeds 100 µs and keeps the CSV as the `qldpc-hil-latency-report` artifact. -Ce test ne doit pas être lancé sur `ubuntu-latest` : il nécessite `/dev/mem`, le contrôleur AXI-DMA, le bitstream chargé, les buffers udmabuf et une carte ZCU111. La vérification FER du fichier HIL est volontairement un point d’extension ; elle doit être remplacée par le calcul réel `H * correction == syndrome` et, si nécessaire, par une vérification d’erreur logique. +This test must not run on `ubuntu-latest`: it needs `/dev/mem`, the AXI-DMA controller, a loaded bitstream, udmabuf buffers and a ZCU111 board. FER verification in the HIL file is deliberately an extension point; it must be replaced by the real `H * correction == syndrome` check and, if needed, a logical-error check. + +See also [`docs/deployment-zcu111.md`](docs/deployment-zcu111.md) and [`vivado/README.md`](vivado/README.md). diff --git a/docs/deployment-zcu111.md b/docs/deployment-zcu111.md index 609606e..14e2f85 100644 --- a/docs/deployment-zcu111.md +++ b/docs/deployment-zcu111.md @@ -1,12 +1,14 @@ -# Déploiement ZCU111 : PetaLinux, FPGA et driver ARM +# ZCU111 deployment: PetaLinux, FPGA and ARM driver (UNRUN) -Cette procédure décrit le déploiement de la chaîne qLDPC sur une carte AMD/Xilinx Zynq UltraScale+ RFSoC ZCU111. Elle suppose que les artefacts PetaLinux, le bitstream FPGA, l’overlay Device Tree et le driver ARM ont déjà été générés. +> **Status: UNRUN.** This procedure has not been executed and no board results are published. It documents the intended flow only. -## 1. Préparer la carte MicroSD +This procedure describes deploying the qLDPC chain to an AMD/Xilinx Zynq UltraScale+ RFSoC ZCU111 board. It assumes that the PetaLinux artifacts, the FPGA bitstream, the Device Tree overlay and the ARM driver have already been generated. -Créer deux partitions : une partition `BOOT` en FAT32 d’au moins 1 Go avec les drapeaux `boot` et `lba`, puis une partition `rootfs` en ext4 avec l’espace restant. +## 1. Prepare the microSD card -Copier les fichiers de démarrage dans la partition FAT32 : +Create two partitions: a FAT32 `BOOT` partition of at least 1 GB with the `boot` and `lba` flags, and an ext4 `rootfs` partition with the remaining space. + +Copy the boot files to the FAT32 partition: ```bash cp BOOT.BIN image.ub boot.scr /media/$USER/BOOT/ @@ -14,17 +16,17 @@ sudo tar -xvf rootfs.tar.gz -C /media/$USER/rootfs/ sync ``` -Configurer le commutateur de démarrage SW6 de la ZCU111 en mode SD : switch 1 `OFF`, switch 2 `ON`, switch 3 `OFF`, switch 4 `OFF`. +Set the ZCU111 boot switch SW6 to SD mode: switch 1 `OFF`, switch 2 `ON`, switch 3 `OFF`, switch 4 `OFF`. -## 2. Démarrer la carte et programmer le FPGA +## 2. Boot the board and program the FPGA -Connecter le port USB-UART, puis ouvrir la console série à 115200 bauds : +Connect the USB-UART port and open the serial console at 115200 baud: ```bash picocom -b 115200 /dev/ttyUSB1 ``` -Pour programmer un bitstream dynamiquement avec FPGA Manager, copier le bitstream et l’overlay dans `/lib/firmware`, puis exécuter : +To program a bitstream dynamically with FPGA Manager, copy the bitstream and overlay to `/lib/firmware`, then run: ```bash mkdir -p /lib/firmware @@ -34,39 +36,39 @@ fpgautil -b /lib/firmware/qldpc_decoder.bit.bin \ -o /lib/firmware/qldpc_overlay.dtbo ``` -La LED DONE de la carte doit confirmer la programmation du FPGA. +The board's DONE LED should confirm that the FPGA is programmed. -## 3. Compiler et exécuter le driver ARM +## 3. Build and run the ARM driver -Transférer le driver sur la carte : +Copy the driver to the board: ```bash -scp axi_dma_driver.cpp root@:/root/ +scp axi_dma_driver.cpp root@:/root/ ``` -Le compiler pour le Cortex-A53 : +Compile it for the Cortex-A53: ```bash g++ -O3 -march=armv8-a -mcpu=cortex-a53 \ axi_dma_driver.cpp -o axi_dma_driver ``` -Vérifier la présence des buffers udmabuf : +Check that the udmabuf buffers exist: ```bash ls -l /dev/udmabuf* ``` -Lancer ensuite le driver avec les privilèges nécessaires : +Then run the driver with the required privileges: ```bash ./axi_dma_driver ``` -## 4. Précautions matérielles +## 4. Hardware precautions -Les adresses AXI-DMA, les canaux, les interruptions, la largeur de données et les adresses physiques des buffers doivent correspondre exactement au Block Design Vivado et au Device Tree déployés. Ne pas utiliser les adresses d’exemple sur une carte différente ou sans vérifier la réservation CMA/udmabuf. +AXI-DMA addresses, channels, interrupts, data width and physical buffer addresses must match the deployed Vivado Block Design and Device Tree exactly. Do not use the example addresses on a different board or without checking the CMA/udmabuf reservation. -Le driver utilise un accès matériel privilégié à `/dev/mem` dans sa version actuelle. Il ne doit pas être exécuté sur un poste de développement ou sur une carte dont la cartographie mémoire n’a pas été validée. La version avec udmabuf doit récupérer les adresses physiques depuis `/sys/class/u-dma-buf/` et mapper les périphériques `/dev/udmabuf_*` plutôt que de dépendre d’adresses physiques codées en dur. +The current driver uses privileged hardware access through `/dev/mem`. Do not run it on a development workstation or on a board whose memory map has not been validated. A udmabuf version should read physical addresses from `/sys/class/u-dma-buf/` and map the `/dev/udmabuf_*` devices instead of relying on hard-coded physical addresses. -La CI GitHub Actions ne peut pas exécuter cette procédure : elle ne dispose ni du matériel ZCU111, ni de Vivado/Vitis HLS, ni des périphériques Linux embarqués nécessaires. Elle continue de valider les tests logiciels, le benchmark et la cohérence des sources. +GitHub Actions CI cannot run this procedure: it has no ZCU111 hardware, no Vivado/Vitis HLS, and none of the required embedded Linux devices. CI continues to validate the software tests, the micro-benchmark and source consistency. diff --git a/drivers/axi_dma_driver.cpp b/drivers/axi_dma_driver.cpp index 1a648e0..b6774d6 100644 --- a/drivers/axi_dma_driver.cpp +++ b/drivers/axi_dma_driver.cpp @@ -6,7 +6,7 @@ #include #include -// À ajuster selon l’adresse réellement attribuée dans le Block Design Vivado. +// Adjust to the address actually assigned in the Vivado Block Design. constexpr uintptr_t AXI_DMA_BASE_ADDR = 0xA0000000; constexpr size_t AXI_DMA_MAP_SIZE = 0x10000; @@ -34,7 +34,7 @@ class AxiDmaDriver { uint8_t* tx_virt_buf_ = nullptr; uint8_t* rx_virt_buf_ = nullptr; - // Ces adresses doivent correspondre à une zone CMA/udmabuf réservée. + // These addresses must match a reserved CMA/udmabuf region. uintptr_t tx_phys_addr_ = 0x10000000; uintptr_t rx_phys_addr_ = 0x10020000; @@ -53,7 +53,7 @@ class AxiDmaDriver { void* mapped = mmap(nullptr, AXI_DMA_MAP_SIZE, PROT_READ | PROT_WRITE, MAP_SHARED, dev_mem_fd_, AXI_DMA_BASE_ADDR); if (mapped == MAP_FAILED) { - std::cerr << "Erreur: échec du mappage des registres AXI-DMA\n"; + std::cerr << "Error: failed to map AXI-DMA registers\n"; return false; } dma_regs_ = reinterpret_cast(mapped); @@ -66,7 +66,7 @@ class AxiDmaDriver { dev_mem_fd_, rx_phys_addr_)); if (tx_virt_buf_ == MAP_FAILED || rx_virt_buf_ == MAP_FAILED) { - std::cerr << "Erreur: échec du mappage des tampons DMA\n"; + std::cerr << "Error: failed to map DMA buffers\n"; return false; } @@ -104,7 +104,7 @@ class AxiDmaDriver { std::chrono::duration_cast(end - start).count(); std::copy(rx_virt_buf_, rx_virt_buf_ + CORRECTION_BYTES, correction_out); #if !defined(AXI_DMA_DRIVER_QUIET) - std::cout << "[ARM Host] Décodage FPGA exécuté en : " << elapsed_us << " us\n"; + std::cout << "[ARM Host] FPGA decode took: " << elapsed_us << " us\n"; #endif } }; diff --git a/hil/hil_benchmark.cpp b/hil/hil_benchmark.cpp index 2810b74..230b4d8 100644 --- a/hil/hil_benchmark.cpp +++ b/hil/hil_benchmark.cpp @@ -34,7 +34,7 @@ double max_latency_us() return 100.0; } -// Remplacer par H * correction == syndrome et le contrôle d’erreur logique. +// Replace with H * correction == syndrome and a logical-error check. bool verify_correction(const uint8_t*, const uint8_t*) { return true; @@ -59,7 +59,7 @@ int main() int logical_errors = 0; std::cout << "[HIL] Lancement de " << num_tests - << " décodages matériels.\n"; + << " hardware decodes.\n"; for (int i = 0; i < num_tests; ++i) { for (auto& byte : syndrome) { byte = static_cast(distribution(generator)); @@ -91,7 +91,7 @@ int main() std::ofstream csv("latencies_report.csv"); if (!csv) { - std::cerr << "Erreur: impossible de créer latencies_report.csv\n"; + std::cerr << "Error: cannot create latencies_report.csv\n"; return 1; } csv << "index,latency_ns\n"; @@ -108,10 +108,10 @@ int main() const double limit = max_latency_us(); if (max_ns / 1000.0 > limit) { - std::cerr << "FAIL: latence maximale supérieure à " << limit + std::cerr << "FAIL: maximum latency above " << limit << " us.\n"; return 2; } - std::cout << "PASS: toutes les latences sont sous " << limit << " us.\n"; + std::cout << "PASS: all latencies are below " << limit << " us.\n"; return 0; } diff --git a/hls/run_hls.tcl b/hls/run_hls.tcl index 688e946..521a562 100644 --- a/hls/run_hls.tcl +++ b/hls/run_hls.tcl @@ -1 +1,28 @@ -# ==============================================================================\n# Script de synthèse automatique AMD Vitis HLS - Décodeur qLDPC\n# ==============================================================================\n\nset script_dir [file dirname [file normalize [info script]]]\n\nopen_project qldpc_hls_project\nset_top qldpc_decode_kernel\n\nadd_files "$script_dir/qldpc_kernel.cpp" -cflags "-std=c++14 -I$script_dir"\nadd_files -tb "$script_dir/testbench.cpp" -cflags "-std=c++14 -I$script_dir"\n\nopen_solution "solution_300mhz" -flow_target vitis\nset_part {xczu28dr-ffvg1517-2-e}\ncreate_clock -period 3.33 -name default\n\nconfig_interface -m_axi_addr64\nconfig_compile -pipeline_loops 1\n\ncsim_design\ncsynth_design\ncosim_design -trace_level all\nexport_design -format ip_catalog \\\n -description "qLDPC BP-OSD-CS Hardware Decoder Kernel" \\\n -vendor "sparkainlp" -version "1.0"\n\nexit\n +# ============================================================================== +# AMD Vitis HLS synthesis script - qLDPC decoder (300 MHz TARGET; UNRUN in CI) +# ============================================================================== + +set script_dir [file dirname [file normalize [info script]]] + +open_project qldpc_hls_project +set_top qldpc_decode_kernel + +add_files "$script_dir/qldpc_kernel.cpp" -cflags "-std=c++14 -I$script_dir" +add_files -tb "$script_dir/testbench.cpp" -cflags "-std=c++14 -I$script_dir" + +open_solution "solution_300mhz" -flow_target vitis +set_part {xczu28dr-ffvg1517-2-e} +create_clock -period 3.33 -name default + +config_interface -m_axi_addr64 +config_compile -pipeline_loops 1 + +csim_design +csynth_design +cosim_design -trace_level all +export_design -format ip_catalog \ + -description "qLDPC BP-OSD-CS Hardware Decoder Kernel" \ + -vendor "sparkainlp" -version "1.0" + +exit + diff --git a/hls/run_hls_400mhz.tcl b/hls/run_hls_400mhz.tcl index da851d1..59d8fe6 100644 --- a/hls/run_hls_400mhz.tcl +++ b/hls/run_hls_400mhz.tcl @@ -1,5 +1,5 @@ # ================================================================================ -# Synthèse Vitis HLS qLDPC - cible 400 MHz +# qLDPC Vitis HLS synthesis - 400 MHz TARGET (UNRUN in CI) # ================================================================================ set script_dir [file dirname [file normalize [info script]]] diff --git a/main.cpp b/main.cpp index ad0207f..9983800 100644 --- a/main.cpp +++ b/main.cpp @@ -4,7 +4,7 @@ int main() { - std::cout << "qldpc_decoder_cpp: bibliothèque ldpc chargée avec succès.\n"; - std::cout << "OpenMP est activé au niveau de la cible CMake.\n"; + std::cout << "qldpc_decoder_cpp: ldpc library loaded.\n"; + std::cout << "OpenMP is enabled on the CMake target.\n"; return 0; } diff --git a/vivado/README.md b/vivado/README.md index 68b3355..27293b7 100644 --- a/vivado/README.md +++ b/vivado/README.md @@ -1,28 +1,30 @@ -# Block Design Vivado RFSoC +# Vivado Block Design for the RFSoC (UNRUN) -Le script [`create_bd.tcl`](./create_bd.tcl) génère le Block Design `system_bd` pour un AMD Zynq UltraScale+ RFSoC ZCU111 (`xczu28dr-ffvg1517-2-e`). Il ajoute le dépôt IP produit par Vitis HLS, instancie le processeur RFSoC, le convertisseur RF, les FIFOs AXI4-Stream et le noyau `qldpc_decode_kernel`. +> **Status: UNRUN.** These scripts have not been run in CI and no timing or implementation report is published. Frequencies below are **TARGET** values. -Le script suppose que le projet Vitis HLS a déjà été généré avec l’IP disponible à l’emplacement suivant : +The script [`create_bd.tcl`](./create_bd.tcl) generates the `system_bd` Block Design for an AMD Zynq UltraScale+ RFSoC ZCU111 (`xczu28dr-ffvg1517-2-e`). It adds the IP repository produced by Vitis HLS and instantiates the RFSoC processing system, the RF data converter, AXI4-Stream FIFOs and the `qldpc_decode_kernel` core. + +The script assumes that the Vitis HLS project has already been generated, with the IP available at: ```text ./qldpc_hls_project/solution_300mhz/impl/ip ``` -Depuis un environnement AMD Vivado correctement configuré, lancer : +From a correctly configured AMD Vivado environment, run: ```bash vivado -mode batch -source vivado/create_bd.tcl ``` -Le script exécute ensuite `validate_bd_design` et sauvegarde le Block Design. La génération dépend des versions installées des IP AMD/Xilinx et ne peut pas être validée dans la CI logicielle standard, qui ne fournit pas Vivado ni les licences FPGA nécessaires. +The script then runs `validate_bd_design` and saves the Block Design. Generation depends on the installed AMD/Xilinx IP versions and cannot be validated in the standard software CI, which provides neither Vivado nor the required FPGA licences. -## Variante 400 MHz +## 400 MHz variant -La variante [`../hls/run_hls_400mhz.tcl`](../hls/run_hls_400mhz.tcl) crée une solution `solution_400mhz` avec une période cible de `2.500 ns`. Le script [`optimize_timing_400mhz.tcl`](./optimize_timing_400mhz.tcl) active le retiming, les stratégies d’implémentation orientées performance et génère `timing_400mhz_report.txt`. +[`../hls/run_hls_400mhz.tcl`](../hls/run_hls_400mhz.tcl) creates a `solution_400mhz` solution with a `2.500 ns` target period. [`optimize_timing_400mhz.tcl`](./optimize_timing_400mhz.tcl) enables retiming and performance-oriented implementation strategies and writes `timing_400mhz_report.txt`. ```bash vitis_hls -f hls/run_hls_400mhz.tcl vivado -mode batch -source vivado/optimize_timing_400mhz.tcl ``` -La cible 400 MHz n’est atteinte que si le rapport post-routage confirme un WNS supérieur ou égal à `0 ns`. Les directives d’optimisation ne constituent pas une garantie de fréquence : la validation dépend du placement-routage réel, de la version des outils, des contraintes d’horloge et de la configuration exacte du Block Design. +The 400 MHz target is met only if the post-route report shows WNS ≥ `0 ns`. The optimisation directives do not guarantee the frequency: closure depends on actual place-and-route, tool version, clock constraints and the exact Block Design configuration. diff --git a/vivado/create_bd.tcl b/vivado/create_bd.tcl index e2e3ff9..c08555b 100644 --- a/vivado/create_bd.tcl +++ b/vivado/create_bd.tcl @@ -1,12 +1,12 @@ # ============================================================================== -# Génération du Block Design Vivado - Zynq UltraScale+ RFSoC & qLDPC +# Vivado Block Design generation - Zynq UltraScale+ RFSoC & qLDPC (UNRUN in CI) # ================================================================================ create_project qldpc_vivado_bd ./qldpc_vivado_bd \ -part xczu28dr-ffvg1517-2-e -force create_bd_design "system_bd" -# Dépôt IP produit par Vitis HLS. +# IP repository produced by Vitis HLS. set_property ip_repo_paths \ ./qldpc_hls_project/solution_300mhz/impl/ip [current_project] update_ip_catalog @@ -21,7 +21,7 @@ apply_bd_automation \ create_bd_cell -type ip -vlnv xilinx.com:ip:usp_rf_data_converter:2.6 \ rf_data_converter -# Noyau qLDPC synthétisé par Vitis HLS. +# qLDPC kernel synthesised by Vitis HLS. create_bd_cell -type ip \ -vlnv sparkainlp:user:qldpc_decode_kernel:1.0 qldpc_decoder @@ -29,7 +29,7 @@ create_bd_cell -type ip \ create_bd_cell -type ip -vlnv xilinx.com:ip:axis_data_fifo:2.0 axis_fifo_in create_bd_cell -type ip -vlnv xilinx.com:ip:axis_data_fifo:2.0 axis_fifo_out -# Horloge à 300 MHz. +# 300 MHz clock (TARGET). create_bd_cell -type ip -vlnv xilinx.com:ip:clk_wiz:6.0 clk_wiz_300MHz set_property -dict [list CONFIG.CLKOUT1_REQUESTED_OUT_FREQ {300.000}] \ [get_bd_cells clk_wiz_300MHz] @@ -37,19 +37,19 @@ connect_bd_net [get_bd_pins zynq_ps/pl_clk0] \ [get_bd_pins clk_wiz_300MHz/clk_in1] set system_clk [get_bd_pins clk_wiz_300MHz/clk_out1] -# Flux syndrome : RF ADC -> FIFO -> décodeur HLS. +# Syndrome stream: RF ADC -> FIFO -> HLS decoder. connect_bd_intf_net [get_bd_intf_pins rf_data_converter/m00_axis] \ [get_bd_intf_pins axis_fifo_in/S_AXIS] connect_bd_intf_net [get_bd_intf_pins axis_fifo_in/M_AXIS] \ [get_bd_intf_pins qldpc_decoder/in_syndrome] -# Flux correction : décodeur HLS -> FIFO -> RF DAC. +# Correction stream: HLS decoder -> FIFO -> RF DAC. connect_bd_intf_net [get_bd_intf_pins qldpc_decoder/out_correction] \ [get_bd_intf_pins axis_fifo_out/S_AXIS] connect_bd_intf_net [get_bd_intf_pins axis_fifo_out/M_AXIS] \ [get_bd_intf_pins rf_data_converter/s00_axis] -# Contrôle AXI-Lite depuis le processeur ARM. +# AXI-Lite control from the ARM processor. apply_bd_automation -rule xilinx.com:bd_rule:axi4 \ -config { Master "/zynq_ps/M_AXI_HPM0_FPD" Clk "Auto" } \ [get_bd_intf_pins qldpc_decoder/s_axi_control] diff --git a/vivado/optimize_timing_400mhz.tcl b/vivado/optimize_timing_400mhz.tcl index 3e7c9a3..5e5b959 100644 --- a/vivado/optimize_timing_400mhz.tcl +++ b/vivado/optimize_timing_400mhz.tcl @@ -1,20 +1,20 @@ # ================================================================================ -# Optimisations avancées de timing Vivado - cible 400 MHz +# Advanced Vivado timing optimisations - 400 MHz TARGET (UNRUN in CI) # ================================================================================ -# Retiming de synthèse. +# Synthesis retiming. set_property STEPS.SYNTH_DESIGN.ARGS.MORE_OPTIONS {-retiming} [get_runs synth_1] -# Stratégies d’implémentation orientées performance. +# Performance-oriented implementation strategies. set_property STRATEGY Performance_ExplorePostRoutePhysOpt [get_runs impl_1] set_property STEPS.PLACE_DESIGN.ARGS.DIRECTIVE Explore [get_runs impl_1] set_property STEPS.ROUTE_DESIGN.ARGS.DIRECTIVE PerformanceExplore [get_runs impl_1] -# Synthèse et implémentation. +# Synthesis and implementation. launch_runs synth_1 -jobs 8 wait_on_run synth_1 launch_runs impl_1 -to_step write_bitstream -jobs 8 wait_on_run impl_1 -# Rapport des chemins critiques et du slack à 400 MHz. +# Critical-path and slack report at 400 MHz. report_timing_summary -max_paths 10 -file timing_400mhz_report.txt