diff --git a/.github/workflows/build-and-measure.yml b/.github/workflows/build-and-measure.yml new file mode 100644 index 0000000..50b85a6 --- /dev/null +++ b/.github/workflows/build-and-measure.yml @@ -0,0 +1,201 @@ +# Build the controller both ways, and measure what recording a slot costs. +# +# The matrix leg is the one thing that has to vary: libe3's stage recorder is a +# build-time decision (-DLIBE3_ENABLE_LATREC), carried to us as a PUBLIC compile +# definition, so "traced" and "untraced" are two different binaries rather than +# two run modes. Building both is what stops the untraced path from rotting +# unnoticed - and the untraced leg is the one that proves the stamps really do +# compile away to nothing. +name: build and measure + +on: + push: + branches: [main] + pull_request: + workflow_dispatch: + +permissions: + contents: read + +jobs: + build: + name: build (latrec=${{ matrix.latrec }}) + runs-on: ubuntu-latest + # Cold, this is asn1c-from-source + libe3 + jbpf and its third-party tree + + # the controller. Comfortably inside an hour; nowhere near the 6h default. + timeout-minutes: 60 + strategy: + fail-fast: false + matrix: + latrec: [on, off] + + steps: + - uses: actions/checkout@v4 + with: + # build.sh fetches the submodules itself and then runs jbpf's + # init_and_patch_submodules.sh over them. Letting the checkout action + # do it first would hand that patch script a tree it did not lay out. + submodules: false + + # The libe3 pin decides which asn1c revision is wanted, so it belongs in + # the cache key. hashFiles cannot read a gitlink, and the submodule is not + # checked out yet, so read the pinned SHA straight out of the tree object. + - name: Read the libe3 pin + id: pin + run: echo "sha=$(git rev-parse HEAD:libe3)" >> "$GITHUB_OUTPUT" + + # asn1c is built from the mouse07410 fork, which is minutes of autotools. + # No restore-keys: a stale asn1c is worse than no cache, because it would + # generate against a different grammar than the pinned libe3 expects. + - name: Cache asn1c + id: cache-asn1c + uses: actions/cache@v4 + with: + path: /opt/asn1c + key: asn1c-${{ runner.os }}-${{ steps.pin.outputs.sha }} + + - name: Cache ccache + uses: actions/cache@v4 + with: + path: ~/.ccache + key: ccache-${{ runner.os }}-latrec${{ matrix.latrec }}-${{ github.sha }} + restore-keys: | + ccache-${{ runner.os }}-latrec${{ matrix.latrec }}- + ccache-${{ runner.os }}- + + - name: Install dependencies and build + env: + # -march=native would tune to whatever CPU this runner happens to be, + # which breaks both ccache reuse across heterogeneous runners and any + # comparison of the measurements below between runs. + E3C_CMAKE_ARGS: -DUSE_NATIVE=OFF + run: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends ccache + export PATH="/usr/lib/ccache:${PATH}" + if [ "${{ matrix.latrec }}" = "on" ]; then + ./build.sh --install-deps --latrec + else + ./build.sh --install-deps + fi + + - name: Record the machine the numbers came from + run: | + { + echo "## Environment (latrec=${{ matrix.latrec }})" + echo + echo '```' + grep -m1 'model name' /proc/cpuinfo || true + echo "nproc: $(nproc)" + echo "/dev/shm: $(df -h /dev/shm | tail -1)" + echo "libe3 pin: $(git -C libe3 rev-parse --short HEAD) ($(cat libe3/VERSION))" + echo '```' + } > report.md + + # The claim being tested: with the recorder off, nothing links against the + # recorder runtime. A timing test cannot show that - only the symbol + # table can. + # + # The check is on global and undefined symbols specifically. libe3 only + # compiles src/core/latrec.c when the recorder is on, so a traced binary + # carries `D latrec_tls` plus `T latrec_seq_next` / `latrec_tls_open_as` + # / `latrec_set_output_dir`; an untraced one carries none of them. What + # an untraced binary can still carry is lowercase `t` entries for the + # inline no-op stubs latrec.h substitutes - local copies the optimiser + # did not bother to discard. Those are empty functions, not the + # recorder, and grepping for them would assert something that is not + # true and would fail for the wrong reason. + - name: Assert nothing links the recorder + if: matrix.latrec == 'off' + run: | + status=0 + for binary in out/bin/e3_controller out/bin/bench_stage_recording; do + # Uppercase type letter = global or undefined, i.e. a real link. + if nm -C "$binary" | grep -E ' [A-Z] ' | grep -i latrec; then + echo "::error::$binary links latrec symbols in an untraced build" + status=1 + else + echo "ok: $binary links no latrec symbols" + fi + if nm -C "$binary" | grep -qw latrec_tls; then + echo "::error::$binary carries the latrec ring registry" + status=1 + fi + done + [ "$status" -eq 0 ] || exit 1 + { + echo + echo '## Not linked' + echo + echo 'Neither binary links any latrec symbol, and neither carries the ring' + echo 'registry. Only latrec.h'"'"'s inline no-op stubs remain, as local symbols.' + } >> report.md + + - name: Measure what recording a slot costs + run: | + mkdir -p rings + { + echo + ./out/bin/bench_stage_recording \ + --latrec-dir "$PWD/rings" \ + --csv-path "$PWD/bench_csv_arm.log" + } >> report.md 2>&1 + # The bench fails if any stamp had to be clamped: that would mean its + # synthetic slot times are not ascending, so its latrec arm would be + # exercising the clamp path rather than the one production takes. + + # Converting the capture with libe3's own tool is what checks the parts of + # the contract that live outside this repository: that the ring role maps + # to the component we intend, that the stage ids form a complete source + # leg, and that the emit-tail hop column materialises. A capture that + # wrapped is not a valid measurement, so fail on it. + - name: Convert and validate the capture + if: matrix.latrec == 'on' + run: | + python3 -m pip install --quiet numpy + TOOLS=/usr/local/share/libe3/tools + python3 "$TOOLS/latrec2csv.py" rings | tee convert.log + { + echo + echo '## Capture' + echo + echo '```' + cat convert.log + echo '```' + } >> report.md + + # The role must land in ocudu.csv, not other.csv: latrec2csv maps the + # role prefix through its own table, so a rename here silently + # reattributes every record to an unknown component. + test -f rings/csv/ocudu.csv \ + || { echo "::error::records were not attributed to the ocudu component"; exit 1; } + + # The emit tail replaces the removed CSV's emit_ns, and only exists + # because (ENCODE_E3SM_DONE -> WAIT_ENTER) is a declared extra hop. + head -2 rings/csv/ocudu.csv | grep -q 'ENCODE_E3SM_DONE__WAIT_ENTER_us' \ + || { echo "::error::the emit-tail hop column is missing"; exit 1; } + + python3 - <<'PY' + import csv, sys + with open('rings/csv/rings.csv') as fh: + rows = list(csv.DictReader(fh)) + if not rows: + sys.exit('::error::no rings were written') + for r in rows: + # A wrapped ring has lost records off the front, so any span + # computed across the wrap is wrong. Better to fail than to + # publish a number from a truncated capture. + if r['wrapped'] != '0' or int(r['lost_records'] or 0): + sys.exit(f"::error::{r['ring']} wrapped ({r['lost_records']} lost) " + "- shorten the run or raise LATREC_ENTRIES_LOG2") + print(f"ok: {r['ring']} {r['rec_count']} records, no wrap, " + f"clock {r['clock_ns_per_call']} ns/call") + PY + + - uses: actions/upload-artifact@v4 + with: + name: report-latrec-${{ matrix.latrec }} + path: | + report.md + rings/csv/ + if-no-files-found: warn diff --git a/CMakeLists.txt b/CMakeLists.txt index 3601ff8..7b7fff8 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -29,6 +29,57 @@ file(MAKE_DIRECTORY ${OUTPUT_DIR}/lib) # Disable tests and examples — we only need the libraries set(BUILD_TESTING OFF CACHE BOOL "Disable jbpf tests" FORCE) +# Stop jbpf from re-initialising its own submodules. +# +# jbpf's CMakeLists.txt runs init_and_patch_submodules.sh when INITIALIZE_SUBMODULES +# is ON, which is its default. That script `rm -rf`s and re-clones jbpf/3p/*, so on +# any tree where git cannot resolve the submodules — a tarball, an `oc cp`/rsync +# copy into a container, a git worktree — it DESTROYS the working 3p checkout and +# then fails: +# +# fatal: not a git repository +# cp: cannot create regular file '3p/ebpf-verifier/': Not a directory +# +# and the tree is left with ebpf-verifier and ubpf gone, so a plain re-configure +# cannot recover. build.sh already populates the submodules itself +# (jbpf/init_and_patch_submodules.sh), so there is nothing to gain from letting +# jbpf redo it. ocudu forces the same value for the same reason +# (ocudu CMakeLists.txt, the ENABLE_JBPF block). +# +# FORCE because it is a cached option in jbpf; without FORCE a stale ON from an +# earlier configure would survive. +set(INITIALIZE_SUBMODULES OFF CACHE BOOL "Let jbpf init its own submodules" FORCE) + +# Fail early and legibly if the submodules are genuinely absent, rather than +# letting the first missing header surface hundreds of lines into the build. +# +# A per-module sentinel FILE, one that module actually ships. +# +# Two traps to avoid here, both hit while testing this: +# - `ck` is autotools (configure/Makefile.in), so CMakeLists.txt is the wrong +# sentinel for it and gives a false alarm. +# - `file(GLOB dir/*)` is NOT a usable emptiness test: unlike a shell glob, +# CMake's matches dotfiles, so a de-populated submodule that still has +# .git/.github looks fully populated. +set(_jbpf_3p_checks + "ubpf/CMakeLists.txt" + "ebpf-verifier/CMakeLists.txt" + "mimalloc/CMakeLists.txt" + "ck/configure") +foreach(_jbpf_3p_file IN LISTS _jbpf_3p_checks) + if(NOT EXISTS "${CMAKE_SOURCE_DIR}/jbpf/3p/${_jbpf_3p_file}") + message(FATAL_ERROR + "jbpf/3p/${_jbpf_3p_file} is missing, so jbpf's third-party submodules are " + "not populated. jbpf's own auto-init is disabled here (see above, it would " + "destroy them), so populate them yourself:\n" + " git submodule update --init --recursive # then\n" + " ./jbpf/init_and_patch_submodules.sh\n" + "or, for a copied tree with no git metadata, copy jbpf/ wholesale from a " + "working checkout.") + endif() +endforeach() +unset(_jbpf_3p_checks) + # Override add_test and set_tests_properties so jbpf subdir doesn't register tests macro(add_test) endmacro() @@ -41,6 +92,21 @@ add_subdirectory(jbpf) unset(add_test) unset(set_tests_properties) +# ---- Codelet verifier (offline) ---- +# The gNB does not verify codelets at load time: jbpf's path is +# ubpf_load_elf_ex + ubpf_compile, and ubpf's JIT emits no memory bounds +# checks. This binary, run from codelets/Makefile at build time, is therefore +# the only memory-safety gate these codelets ever pass through. +# +# It needs ocudu's janus headers, which define the hook context structs — the +# verifier derives its ctx descriptors from them with offsetof() so the +# contract has a single source of truth. Point OCUDU_DIR elsewhere if your +# checkout is not a sibling of this one. +set(OCUDU_DIR "${CMAKE_SOURCE_DIR}/../ocudu-e3" + CACHE PATH "Path to the ocudu-e3 checkout (hook context contract)") +set(OCUDU_JANUS_INC "${OCUDU_DIR}/include/ocudu/janus") +add_subdirectory(codelets/verifier) + # ---- ASN.1 (E3 Service Model) ---- add_subdirectory(src/e3sm/asn) @@ -55,6 +121,27 @@ FetchContent_Declare( set(EXPECTED_BUILD_TESTS OFF CACHE BOOL "" FORCE) FetchContent_MakeAvailable(tl_expected) +# ---- yaml-cpp (E3Controller configuration) ---- +# The controller is configured by a single YAML file rather than 20 CLI options. +# YAML specifically (not JSON, which would be free via libe3's nlohmann) because +# ocudu's own configuration is YAML, including the `jbpf:` section this file must +# agree with — keeping both sides in the same language makes them diffable and +# reviewable together, which is the point when the invariant is "these must match". +# +# FetchContent rather than a system package, matching the tl::expected pattern +# above, so the build stays self-contained. +FetchContent_Declare( + yaml_cpp + GIT_REPOSITORY https://github.com/jbeder/yaml-cpp.git + GIT_TAG 0.8.0 + GIT_SHALLOW TRUE +) +set(YAML_CPP_BUILD_TESTS OFF CACHE BOOL "" FORCE) +set(YAML_CPP_BUILD_TOOLS OFF CACHE BOOL "" FORCE) +set(YAML_CPP_BUILD_CONTRIB OFF CACHE BOOL "" FORCE) +set(YAML_CPP_INSTALL OFF CACHE BOOL "" FORCE) +FetchContent_MakeAvailable(yaml_cpp) + # ---- libe3 (E3 agent library) ---- # libe3 is the `libe3/` git submodule (https://github.com/wineslab/libe3.git, # pinned to tag 0.0.4). build.sh builds it with BOTH encoders and installs it to @@ -66,6 +153,36 @@ FetchContent_MakeAvailable(tl_expected) # cmake --build libe3/build -j && sudo cmake --install libe3/build find_package(libe3 REQUIRED) +# ---- libe3 staleness stamp ---- +# +# libe3::libe3 is STATIC IMPORTED (see /usr/local/lib/cmake/libe3/libe3Targets.cmake), +# so this binary carries a SNAPSHOT of liblibe3.a taken at link time. Reinstalling +# libe3 afterwards changes nothing until the controller is relinked, and there is +# no runtime symptom: a changed SLEEP_DURATION or E3AP encoder is simply absent. +# Record what we linked against so the binary can say so on startup and warn when +# /usr/local has moved on. Costs one stat() at boot. +get_filename_component(LIBE3_PREFIX "${libe3_DIR}/../../.." ABSOLUTE) +set(LIBE3_STAMP_LIB "${LIBE3_PREFIX}/lib/liblibe3.a") +set(LIBE3_STAMP_VERSION "${libe3_VERSION}") +set(LIBE3_STAMP_BUILDTYPE "${CMAKE_BUILD_TYPE}") + +# Bake the queue's poll period: it is the value most often changed by hand and +# the one whose staleness is least visible (it only shows up as a shifted +# outbound queue-wait distribution). +set(LIBE3_STAMP_SLEEP_US "unknown") +if(EXISTS "${LIBE3_PREFIX}/include/libe3/lockfree_queue.hpp") + file(STRINGS "${LIBE3_PREFIX}/include/libe3/lockfree_queue.hpp" + _libe3_sleep_line REGEX "SLEEP_DURATION[ \t]*=") + string(REGEX MATCH "microseconds\\(([0-9]+)\\)" _m "${_libe3_sleep_line}") + if(CMAKE_MATCH_1) + set(LIBE3_STAMP_SLEEP_US "${CMAKE_MATCH_1}") + endif() +endif() +message(STATUS "libe3: ${LIBE3_STAMP_VERSION} from ${LIBE3_PREFIX} " + "(static, SLEEP_DURATION=${LIBE3_STAMP_SLEEP_US}us)") +configure_file("${CMAKE_CURRENT_SOURCE_DIR}/cmake/libe3_stamp.h.in" + "${CMAKE_BINARY_DIR}/generated/libe3_stamp.h" @ONLY) + # ---- E3Controller executable ---- add_executable(e3_controller src/e3_controller.cpp @@ -79,6 +196,8 @@ add_executable(e3_controller # shared data plane src/e3sm/iq_pipeline.cpp src/e3sm/slot_iq_pipeline.cpp + # configuration (YAML) + src/e3_config.cpp # writers + decompression src/e3sm/utils/e3sm_shm_writer.cpp src/e3sm/utils/bfp_decompress.cpp @@ -86,6 +205,7 @@ add_executable(e3_controller # Include paths: our own headers + jbpf headers target_include_directories(e3_controller PRIVATE + ${CMAKE_BINARY_DIR}/generated # libe3_stamp.h ${CMAKE_SOURCE_DIR}/include ${CMAKE_SOURCE_DIR}/src ${CMAKE_SOURCE_DIR}/jbpf/src/io @@ -102,6 +222,7 @@ target_link_libraries(e3_controller PRIVATE jbpf::lcm_ipc_lib e3sm_asn libe3::libe3 + yaml-cpp pthread rt dl @@ -111,3 +232,27 @@ target_link_libraries(e3_controller PRIVATE set_target_properties(e3_controller PROPERTIES RUNTIME_OUTPUT_DIRECTORY "${OUTPUT_DIR}/bin" ) + +# ---- Recording-cost bench ---- +# Prices the stage recorder that replaced the per-slot stats CSV. Deliberately +# minimal: it links the same trace header the slot handler uses and nothing +# else -- no jbpf, no Service Model, no shared-memory writer, no encoders. That +# is enough to measure the recording mechanism, and it keeps the target +# buildable (and CI-runnable) with no RAN attached. +# +# There is no "real slot work" arm to link for: under the default +# shm.writer: gnb the gNB converts and writes the row itself, so the controller +# moves no slot data at all. See bench/bench_stage_recording.cpp. +option(E3C_BUILD_BENCH "Build the stage-recording cost bench" ON) + +if(E3C_BUILD_BENCH) + add_executable(bench_stage_recording bench/bench_stage_recording.cpp) + target_include_directories(bench_stage_recording PRIVATE + ${CMAKE_SOURCE_DIR}/include + ${CMAKE_SOURCE_DIR}/src + ) + target_link_libraries(bench_stage_recording PRIVATE libe3::libe3 pthread rt) + set_target_properties(bench_stage_recording PROPERTIES + RUNTIME_OUTPUT_DIRECTORY "${OUTPUT_DIR}/bin" + ) +endif() diff --git a/README.md b/README.md index fb5c547..bcbfd92 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,9 @@ A standalone C++ daemon that bridges ocudu's jbpf shared memory (IPC primary) with the E3 protocol via [libe3](https://github.com/wineslab/libe3). It receives I/Q sample data from jbpf codelets and exposes it as E3 Service Model indications to subscribed dApps. -The controller serves **one** wire encoding at a time (selected with `--encoding`), over a configurable link layer (`--link-layer`) and transport (`--transport`). +The controller serves **one** wire encoding at a time, over a configurable link layer and +transport. All of that — along with the radio geometry, SHM layout, jbpf IPC settings and +thread pinning — lives in a single YAML configuration file; see [Usage](#usage). > To use this E3Controller you need to build and run [this version](https://github.com/wineslab/ocudu-e3) of OCUDU. @@ -21,7 +23,8 @@ cd E3Controller ./build.sh ``` -The binary lands at `out/bin/e3_controller`. +The binary lands at `out/bin/e3_controller`, and the offline codelet verifier at +`out/bin/e3_verifier_cli`. ### `--install-deps` (or `-d`) @@ -53,33 +56,49 @@ packages by hand and re-run `./build.sh` without `--install-deps`. Whether or not `--install-deps` was used, `build.sh` then runs: -1. `git submodule update --init --recursive` (fetches libe3 @ tag `0.0.6` and - jbpf). -2. `jbpf/init_and_patch_submodules.sh` to bring in jbpf's third-party - dependencies. +1. `git submodule update --init --recursive` (fetches libe3 and jbpf at their + pinned commits). +2. `jbpf/init_and_patch_submodules.sh` for jbpf's third-party dependencies, then + `apply_jbpf_patches.sh` for the patches to jbpf's own core. 3. Configure + build libe3 with both encoders (`-DLIBE3_ENABLE_ASN1=ON -DLIBE3_ENABLE_JSON=ON -DLIBE3_BUILD_EXAMPLES=OFF - -DLIBE3_BUILD_TESTS=OFF`), stage `asn1c`'s `BOOLEAN.*` skeletons into - `libe3/build/messages/` (toolchain shim — libe3's E3AP grammar does not use - `BOOLEAN` and the `mouse07410` fork skips them, so we supply the reference - copies), then `sudo cmake --install libe3/build` to `/usr/local`. -4. Configure + build the E3Controller; jbpf is compiled in-tree via - `add_subdirectory`. + -DLIBE3_BUILD_TESTS=OFF`), then `sudo cmake --install libe3/build` to + `/usr/local`. +4. Configure + build the E3Controller and the recording-cost bench; jbpf is + compiled in-tree via `add_subdirectory`. -The `--encoding` flag on the resulting binary is a pure runtime choice because -libe3 is built with both encoders. +`e3.encoding` in the config is therefore a pure runtime choice, because libe3 is built +with both encoders. + +### `--latrec` + +Builds libe3 with `-DLIBE3_ENABLE_LATREC=ON`, compiling in the stage recorder — +see [Stage records](#stage-records). Off by default, and off is the normal way +to run: libe3 only compiles the recorder in when it is set, so without it the +controller's stamps degrade to `latrec.h`'s inline no-ops and nothing links the +recorder runtime. + +There is only this one flag. `LIBE3_ENABLE_LATREC` is a `PUBLIC` compile +definition on `libe3::libe3`, so a traced libe3 turns the controller's stamps on +by itself — nothing has to be passed twice, and a mismatch between the two is a +link error rather than a silently untraced build. Set `LATREC_DEFAULT_DIR=` +alongside it to change the compiled-in default ring directory (otherwise +`/tmp/latrec`). + +`rm -rf libe3/build` before switching `--latrec` on or off: it changes the +behaviour of an installed public header. ### Overrides & re-runs - `JOBS=N ./build.sh` — parallelism (defaults to `nproc`). -- `ASN1C_SKELETON_DIR= ./build.sh` — override where `BOOLEAN.*` are - copied from. Default probe order is - `/opt/asn1c/share/asn1c` → `/usr/local/share/asn1c` → `/usr/share/asn1c`. +- `E3C_CMAKE_ARGS='-DUSE_NATIVE=OFF' ./build.sh` — extra configure flags for + the controller. CI passes exactly this: `-march=native` is right on a + deployment host but wrong for a measurement run on a shared runner, where it + makes numbers incomparable between runs. +- `LIBE3_BUILD_TYPE=RelWithDebInfo ./build.sh` — libe3's build type. - If you re-build without `--install-deps` on a host where the earlier run put `asn1c` under `/opt/asn1c/`, the script re-adds `/opt/asn1c/bin` to `PATH` automatically before invoking `cmake`. -- `rm -rf libe3/build` before re-running is enough to force the `BOOLEAN.*` - shim to re-stage; `cmake` picks the rest up incrementally. ### ASN.1 Code Generation @@ -95,58 +114,361 @@ Generated files go into `build/asn1c_generated/` and are **not** tracked in git. > **Important:** E3Controller (IPC primary) must start **before** ocudu (IPC secondary). ```bash -./out/bin/e3_controller [options] +./out/bin/e3_controller --config configs/e3_controller.yaml +``` + +`--config` is the **only** argument. It replaced 20 command-line options, deliberately: +two configuration mechanisms invite the two to disagree, and the radio geometry has to be +stated in exactly one place. + +A fully documented example ships at [`configs/e3_controller.yaml`](configs/e3_controller.yaml). +Unknown keys are a **startup error**, not a warning — a typo'd `nof_port:` that was +silently ignored would leave the controller sizing rows for the default while you believed +you had configured something else. + +### Example + +100 MHz / 30 kHz SCS, 4x4, ASN.1 encoding, three pinned cores: + +```yaml +# radio geometry — MUST match the running gNB (validated against it at runtime) +radio: + nof_ports: 4 # UL antenna ports + nof_prbs: 273 # 100 MHz @ 30 kHz SCS + nof_symbols: 14 # per slot + scs_khz: 30 # also fixes slots/frame (20) + +# /e3_ran_buffers — owned by the controller, read by the dApp +shm: + name: /e3_ran_buffers + size_bytes: 1073741824 # 1 GiB + cbf16_scale: 1.0 # bf16 -> fp16 scale; must be > 0 + writer: controller # controller | gnb (exactly one may write) + +# must agree with the gNB's own `jbpf:` section +jbpf: + ipc_name: e3_controller + run_path: /dev/shm + mem_size_bytes: 1073741824 + lcm_socket_path: /tmp/jbpf/jbpf_lcm_ipc + codelet_base_path: /workspace/e3_release/E3Controller/codelets + +e3: + encoding: asn1 # asn1 | json + link_layer: zmq # zmq | posix + transport: tcp # tcp | ipc | sctp + setup_port: 9990 + publisher_port: 9991 + subscriber_port: 9999 + +threads: + poll_core: 2 + worker_core: 3 + publisher_core: 4 + poll_interval_us: 100 # ignored when poll_core >= 0 (busy-poll) + +logging: + drops_log_path: "" # empty disables the drop-accounting CSV + latrec_dir: "" # where stage-record rings go; needs --latrec + +target_slot: -1 # -1 = every UL slot +``` + +For a **JSON / cuBB dApp** (such as `adaptive_cpu`), change the `e3:` section — the port +convention differs, so keep one config file per encoding rather than trying to override +individual values at launch: + +```yaml +e3: + encoding: json + link_layer: zmq + transport: tcp + setup_port: 5555 + publisher_port: 5556 + subscriber_port: 5557 ``` -### Options +For a **2x2** deployment, only `radio.nof_ports` changes — the row stride, header and all +derived sizes follow from it, with no recompile: + +```yaml +radio: + nof_ports: 2 + nof_prbs: 273 + nof_symbols: 14 + scs_khz: 30 +``` -| Option | Default | Description | +### Configuration + +| Section | Keys | Notes | |---|---|---| -| `--ipc-name ` | `e3_controller` | IPC shared memory segment name | -| `--run-path ` | `/dev/shm` | jbpf run path | -| `--mem-size ` | `1073741824` (1GB) | Shared memory size | -| `--poll-interval ` | `100` | Poll interval in microseconds (ignored if `--poll-core` is set) | -| `--poll-core ` | `-1` | Pin the polling thread to `` and busy-poll | -| `--worker-core ` | `-1` | Pin the SM worker (decompress/encode/emit) to `` | -| `--publisher-core ` | `-1` | Pin libe3's RAN outbound thread (encode + ZMQ send) to `` (not supported) | -| `--num-prbs ` | `106` | Expected number of PRBs per OFDM symbol (used to filter out PRACH/SRS/control symbols with different PRB counts) | -| `--lcm-socket ` | `/tmp/jbpf/jbpf_lcm_ipc` | LCM IPC socket for codelet loading | -| `--codelet-path ` | (none) | Base directory for codelet binaries (enables auto-loading) | -| `--encoding ` | `asn1` | Wire encoding for the E3 channel: `asn1` or `json`. Runtime-switchable when libe3 is built with both encoders. | -| `--link-layer ` | `zmq` | Link layer: `zmq` or `posix` | -| `--transport ` | `tcp` | Transport: `tcp`, `ipc`, or `sctp` | -| `--setup-port

` | `9990` | E3 channel setup REP port | -| `--publisher-port

` | `9991` | E3 channel indication PUB port | -| `--subscriber-port

` | `9999` | E3 channel control SUB port | -| `--shm-name ` | `/e3_ran_buffers` | POSIX SHM name for IQ data | -| `--shm-size ` | `1073741824` (1GiB) | POSIX SHM size | -| `--target-slot ` | `-1` | Forward ONLY UL slot N (absolute slot 0..19, 30 kHz SCS); `-1` forwards every UL slot | -| `--stats-log ` | (disabled) | Write the per-slot RAN-side stage CSV (gnb/codelet/dispatch/handler + shm/encode/emit) | -| `--pub-stages-log ` | (disabled) | Write libe3's per-PDU publisher-stage CSV (queue_us/encode_us/zmq_send_us/t_sent_us) | -| `--help` | | Show help | +| `radio` | `nof_ports`, `nof_prbs`, `nof_symbols`, `scs_khz` | **Must match the running gNB.** Validated against the RAN — see below. | +| `shm` | `name`, `size_bytes`, `cbf16_scale`, `writer` | The `/e3_ran_buffers` region the controller owns and the dApp reads; `writer` picks which process converts and writes rows (see below) | +| `jbpf` | `ipc_name`, `run_path`, `mem_size_bytes`, `lcm_socket_path`, `codelet_base_path` | Must agree with the gNB's own `jbpf:` YAML section | +| `e3` | `encoding`, `link_layer`, `transport`, `setup_port`, `publisher_port`, `subscriber_port` | `encoding` is `asn1` or `json`; JSON/cuBB dApps expect ports 5555/5556/5557 | +| `threads` | `poll_core`, `worker_core`, `publisher_core`, `poll_interval_us` | `-1` = no pinning (the poll thread then sleeps rather than busy-spinning) | +| `logging` | `drops_log_path`, `latrec_dir` | Drop accounting, and where stage-record rings land ([Stage records](#stage-records)) | +| *(top level)* | `target_slot` | Forward only this slot index within a 10 ms frame; `-1` forwards every UL slot | + +#### Who writes the rows (`shm.writer`) + +| Value | Data path | Requires | +|---|---|---| +| `controller` (default) | codelet copies the grid into the jbpf ring -> the controller converts cbf16 -> fp16 into `/e3_ran_buffers` | any gNB | +| `gnb` | codelet calls the gNB-side publish helper, which converts and writes the row in the PHY RX thread; the jbpf ring carries only a ~64 B descriptor | a gNB with the E3 helpers registered and the descriptor-emitting codelet | + +`gnb` halves total memory traffic (3 MB -> 1.5 MB per slot) by fusing the copy and +the conversion into one pass, and takes the controller out of the data plane entirely. + +**Exactly one process may write.** Both writers keep their own ring cursor, so if both +were active they would overwrite each other's rows with no error anywhere. The switch makes +that structurally impossible: in `gnb` mode the controller's `publish_row_cbf16` +hard-refuses and the row indices come from the RAN instead. + +Either way the controller **owns** the region — it creates, sizes, zero-fills, headers and +tears it down; the helper only attaches (`O_RDWR` without `O_CREAT`, so it cannot race the +owner into creating one with the wrong shape). + +#### Radio geometry is checked, not trusted + +`radio:` has to be declared up front because the SHM region must exist and be sized before +the codelet can be loaded. But the RAN is the real source of truth: the per-slot hook +context carries `nof_ports` / `nof_symbols` / `nof_subcarriers`, so the controller compares +your configuration against the first slot it receives and **refuses to publish on a +mismatch**, naming both sides. + +The two directions are not symmetric, which is why the check exists: + +- Declaring **fewer** ports than the gNB sends is already safe — the gNB-side publish + helper refuses the oversized slot and the codelet reports it. +- Declaring **more** is what needs catching: the surplus antennas are written as silence, + and the dApp cannot distinguish that from a genuinely quiet antenna. A plausible-looking + wrong spectrum, with nothing to notice. + +> **`E3_CBF16_SCALE` no longer has any effect.** The bf16 -> fp16 scale moved into +> `shm.cbf16_scale`, because the gNB-side publish helper needs the *same* value and a +> disagreement would produce different rows with no error at all — bf16 and fp16 are both +> 2 bytes, so a wrong scale is silently wrong data rather than a failure. > **Encoding vs. libe3 build.** When libe3 is built with both encoders > (the recommended build — see [libe3 (git submodule)](#libe3-git-submodule)), -> `--encoding` is a pure runtime choice. If libe3 was built with only one -> encoder, `--encoding` must match it, or the outbound encoder rejects every PDU. +> `e3.encoding` is a pure runtime choice. If libe3 was built with only one +> encoder, it must match, or the outbound encoder rejects every PDU. + +#### Drop accounting + +Off by default; enabled by setting `logging.drops_log_path`. One cumulative row +per second: `uptime_s, published, dropped_total, latrec_clamped`, then a column +per drop reason. Aggregate and throttled, so it does no work on the slot path. + +The drop *counters* are always on — the throttled stderr line and the shutdown +summary do not depend on this path. + +Per-slot stage timing is not here; see [Stage records](#stage-records). + +An example launcher is available [here](start_e3controller_example.sh). + +## Stage records + +Where the time goes on the slot path, recorded without perturbing it. -#### Timing logs +The controller stamps its own stages into **latrec**, the per-thread lock-free +ring recorder libe3 ships in `libe3/latrec.h`. A stamp is one +`clock_gettime(CLOCK_MONOTONIC)` plus four stores into an mmap-backed ring: no +syscall, no allocation, no formatting, no lock and no I/O on the slot path. +Conversion to tables happens offline, out of process, against the ring files. -Both timing logs are **off by default** and enabled by passing a path: +This replaced a per-slot CSV that opened an `ofstream`, wrote a row and flushed +it once per slot inside the sample handler — so the numbers it produced included +the cost of producing them. See [What recording costs](#what-recording-costs). + +Build with `./build.sh --latrec`, then point the rings somewhere with +`logging.latrec_dir`. + +### The stages + +One row per slot, on the forward leg. Box numbering is libe3's +`docs/path-a-e3-loop.md`; the identifiers are the shared catalog's, which names +*operations* rather than components — there is no controller-specific stage +block, and which component performed an operation is read off the ring that +recorded it. + +| Box | Segment | Covers | +|---|---|---| +| A1 | `RECORD_BEGIN` → `PROCESS_BEGIN` | the data recording: jbpf dispatch, then the cbf16 → fp16 convert of the grid and the row write into `/e3_ran_buffers` | +| A2 | `PROCESS_BEGIN` → `ENCODE_E3SM_BEGIN` | getting it to the Service Model: jbpf ring transit, dispatcher poll, queue wait | +| A3 | `ENCODE_E3SM_BEGIN` → `ENCODE_E3SM_DONE` | the E3SM payload encoder | +| — | `ENCODE_E3SM_DONE` → `WAIT_ENTER` | the emit tail, over every subscriber | + +**A1 and A2 are recorded here, and the gNB needs no instrumentation for it.** +Both of A1's boundaries are already on the wire by the time the slot arrives: +`gnb_ts_ns` is stamped on the last symbol with the resource grid complete and +nothing yet copied, and `codelet_ts_ns` just before the codelet submits. So the +controller replays them rather than the RAN keeping a ring of its own. + +That works because the whole chain reads `CLOCK_MONOTONIC` — the gNB hook, +`jbpf_time_get_ns()` (see `jbpf_patches/jbpf_monotonic_time.patch`) and the +dispatcher poll — which is latrec's own clock. No domain conversion, no offset +arithmetic, no rate skew to absorb. **All of those sites have to agree**; if one +drifts back to `CLOCK_REALTIME` the stage intervals silently mix epochs. + +Four sub-hops stay recoverable from `aux` payloads, which is what keeps jbpf +dispatch separable from the data movement: + +| Sub-hop | Value | +|---|---| +| jbpf dispatch | `RECORD_BEGIN.aux2` − `RECORD_BEGIN` | +| convert + row write | `PROCESS_BEGIN` − `RECORD_BEGIN.aux2` | +| jbpf ring transit | `PROCESS_BEGIN.aux` − `PROCESS_BEGIN` | +| queue wait | `ENCODE_E3SM_BEGIN` − `PROCESS_BEGIN.aux` | + +Everything past the emit boundary — E3AP encode, the outbound queue, the +connector send — is libe3's box and is stamped inside libe3. **E3SM and E3AP are +separate boxes**: the Service Model codec is ours, the E3AP codec is the +library's, and keeping them apart is what makes the library's cost separable +from ours. We do not re-time the library's stages; we join to them. + +Slots that never reach the traced region are not stamped. In particular the "no +subscribers" path returns before the first stamp: that is the normal idle state, +and recording it would fill the ring with slots nobody asked for. A slot whose +encode fails closes its row with `SKIPPED` / `LATREC_SKIP_ENCODE`. + +### Back-dating and the clamp + +A1's boundaries happened before the handler ran, so those two stamps carry times +earlier than the moment they are issued. A ring is a single-writer log whose +`t_ns` must ascend, and libe3's reader treats the *one* permitted descent as the +wrap point and silently rotates there — a descent would not raise an error, it +would produce a plausible capture cut at the wrong offset. + +Ordering normally holds with a wide margin: the jbpf hook is a synchronous +inline call, so slot N's codelet has returned before slot N+1 is stamped, and A1 +is tens of microseconds against a slot spacing of at least 500 µs at 30 kHz SCS. +Two things can still break it — a pipeline stall longer than the slot spacing, +and several RU receive threads (one per sector) whose slots interleave on one +ring. So the floor is enforced rather than assumed, and the count of enforced +stamps is reported in `latrec_clamped` in the drop CSV and in the shutdown +summary. **A non-zero count means A1 is understated for that many slots.** + +### Reading a capture + +```bash +python3 /usr/local/share/libe3/tools/latrec2csv.py +``` -- `--stats-log ` — written by `E3SMLayer1` (controller side). One row per - published UL slot: `slot_seq, gnb_to_codelet_us, codelet_to_dispatch_us, - dispatch_to_handler_us, shm_ns, encode_ns, emit_ns, nof_subc, iq_bytes`. -- `--pub-stages-log ` — written by libe3's RAN outbound loop. One row per - SM-emitted PDU: `message_id, queue_us, encode_us, zmq_send_us, t_sent_us`. - (Plumbed into `E3Config.pub_stages_log_path`; libe3 also honours the - `LIBE3_PUB_STAGES_LOG_PATH` env var as a fallback.) +Records land in `ocudu.csv` — the ring role is `e3controller.l1_kpm`, and +`latrec2csv.py` maps the `e3controller` prefix onto the `ocudu` component. The +hops appear as `RECORD_BEGIN__PROCESS_BEGIN_us` and so on, with the emit tail as +`ENCODE_E3SM_DONE__WAIT_ENTER_us`. -The two join end-to-end by `message_id`: `statistics`'s `emit_ns` is the SM-side -enqueue cost and `pub-stages`'s `queue_us`/`encode_us`/`zmq_send_us` pick up -where it leaves off. +**Joining to libe3's records.** The controller publishes each slot's record +sequence with `latrec_ctx_set()` immediately before entering the library; libe3 +stamps it into `EMIT_ENTER`'s `aux`, which surfaces as the `origin_seq` column +on the outbound leg. So: -An example on how to run it it's available [here](start_e3controller_example.sh). +- `source.seq == outbound.origin_seq` links a slot to its emission. Both legs + are in `ocudu.csv`, because the emit boundary and the enqueue are stamped on + the calling thread — ours. +- That outbound row's `seq` then links into `libe3.csv`, where the library's own + outbound thread stamped `DEQUEUE` → `ENCODE_E3AP_DONE` → `SEND_DONE`. + +Two steps, because the outbound leg is split across the two threads that perform +it. There is no shared message identifier doing this work: E3AP's `message_id` +wraps at 1000, and while libe3 ≥ 0.1.2 surfaces it to a *dApp* calling +`send_control`/`send_report`, it is still not visible on the Service Model emit +path this SM uses. + +### Ring sizing and the capture window + +Rings default to 2^18 records (8 MiB per thread). The slot path writes five +records per published slot, so at 30 kHz SCS with every UL slot forwarded +(~2000 slots/s) a default ring holds about **26 seconds** before it wraps: + +```bash +LATREC_ENTRIES_LOG2_E3CONTROLLER_L1_KPM=24 # 512 MiB, ~28 min; ceiling 2^28 +``` + +The variable is the ring role uppercased with non-alphanumerics replaced by `_`. +It sizes the ring — it does not enable anything. + +**A wrapped capture is not a valid measurement**: records are lost off the front +and any span computed across the wrap is wrong. `rings.csv` reports `wrapped` +and `lost_records` for exactly this reason, and CI fails on a non-zero value. + +### What recording costs + +`out/bin/bench_stage_recording` prices the recorder. It links the same trace +header the slot handler uses and nothing else — no jbpf, no Service Model, no +shared-memory writer — so it runs in CI with no RAN attached. + +There is deliberately no "real slot work" arm: under the default +`shm.writer: gnb` the gNB converts and writes the row itself, so the controller +moves no slot data at all, and against real work the stamps were never +resolvable anyway. + +Median of 15 batches, net of the loop floor, `-DUSE_NATIVE=OFF`: + +| Recording one slot | Workstation | CI (EPYC 7763) | Of a 500 µs slot | +|---|---|---|---| +| the removed per-slot CSV | ~1840 ns | ~1485 ns | 0.30–0.37% | +| latrec, 5 records | ~111 ns | ~102 ns | 0.02% | +| **ratio** | **~17×** | **~15×** | | + +Per record that is ~20–22 ns against a measured 22–28 ns clock read: the clock +and essentially nothing else. + +**Compare within a host, never across one.** The two CI legs are separate jobs +and land on whatever runner they get — on one run an EPYC 7763 and an Intel Xeon +6973P-C, where the `csv` arm alone differed by 2.7×. That arm is the calibration +anchor: it is the same code in both builds, so when it disagrees, nothing else +in those two reports is comparable either. + +The untraced leg is further apart still, and for a second reason: with the +recorder off `latrec_tnow()` compiles to `return 0`, so the bench's per-slot +clock read disappears and the loop floor collapses (~1 ns instead of ~30 ns). +Read that leg for the `csv` arm and the not-linked assertion, not for a +comparison against the traced one. + +## Codelets + +The jbpf codelets are built and verified **in this repository**, under +[`codelets/`](codelets/) — there is no longer any dependency on an external SDK tree. + +```bash +cd codelets +make # build + verify every codelet +make verify # re-verify existing objects +make show-config # resolved paths, [ok]/[MISSING] per include root +``` + +Requires `clang` with the BPF target, and `e3_verifier_cli` from the main build. The +context contract (`jbpf_srsran_contexts.h`) is imported from the ocudu checkout rather than +copied — point `OCUDU_DIR` at it if it is not a sibling of this repo: + +```bash +make OCUDU_DIR=/path/to/ocudu-e3 +``` + +`make check-contract` fails if a local copy has drifted from ocudu's. + +### Verification is mandatory + +`make` **fails** if a codelet does not verify, and that is not belt-and-braces: the gNB +does no verification at load time. jbpf's load path is `ubpf_load_elf_ex` followed by +`ubpf_compile`, PREVAIL is not on it, and ubpf's JIT emits no memory bounds checks — only +its interpreter does, and jbpf uses the JIT. So this build step is the only memory-safety +gate these codelets ever pass through. + +`codelets/verifier/e3_verifier_cli.cpp` registers the ocudu program types and the E3 helper +prototypes on top of jbpf's built-ins, deriving its context descriptors with `offsetof` +from ocudu's own header so the contract has one source of truth. It replaces the SDK +image's `srsran_verifier_cli`, which models only the 27-byte `jbpf_ran_ofh_ctx` and +therefore cannot verify the per-slot codelet at all. + +Helper and program-type IDs live in [`codelets/include/jbpf_e3_ids.h`](codelets/include/jbpf_e3_ids.h), +shared by the codelets, the verifier, and the gNB-side helper registration. If those three +disagree, a codelet that verifies cleanly still fails to load. ## Architecture @@ -294,7 +616,7 @@ wineslab's own and is not part of NVIDIA's schema. ### Single encoding per process The controller serves exactly **one** wire encoding at a time, fixed at startup -by `--encoding` (libe3 is built with both encoders, so this is a runtime choice +by `e3.encoding` (libe3 is built with both encoders, so this is a runtime choice — see [libe3 (git submodule)](#libe3-git-submodule)). To serve both an ASN.1 dApp and a JSON dApp simultaneously, run two controller instances on different port triples. This is the deliberate simplification from the earlier dual-channel design: with @@ -303,8 +625,8 @@ fan-out simply read `E3Agent::config().encoding` — no per-dApp encoding lookup no race during simultaneous setup. Note that **RF=1 Spectrum is ASN.1/APER-only** — there is no JSON encoder for its -in-band IQ indication, so it is only useful under `--encoding asn1` (under -`--encoding json` the SM warns once and drops indications). **RF=2 L1-KPM** +in-band IQ indication, so it is only useful under `e3.encoding: asn1` (under +`json` the SM warns once and drops indications). **RF=2 L1-KPM** supports both encodings. ### PRB blacklist control is a stub @@ -318,5 +640,35 @@ the blacklist to the RAN scheduler is not implemented. The marks the spot. dApps that rely on the control side-effect (rather than just the ACK) will not see scheduler behaviour change. -### Codelet Verifier -Missing codelet verifier. This is planned as future work and will provide a framework to build and verify codelets, enabling developers to safely extend the E3Controller functionality attached to the various hooks available in OCUDU. +### Slot buffer size is compile-time in the codelet + +The radio geometry is runtime configuration on the controller side (`radio:` in the YAML), +but `MAX_SLOT_IQ_BYTES` in +[`codelets/uplink_slot_samples/uplink_slot_data.h`](codelets/uplink_slot_samples/uplink_slot_data.h) +is still a compile-time constant — currently `733824`, i.e. 4 ports x 14 symbols x 3276 +subcarriers x 4 bytes. + +This one is **not** an oversight and cannot be made runtime: it sizes the codelet's +jbpf output-map struct, and the eBPF verifier requires compile-time-known struct sizes. +`src/e3sm/slot_iq_pipeline.h` includes that header, so the controller's view of the sample +is sized by it too. + +Consequence: raising the antenna count **above** what the constant covers (e.g. 8x8) needs +the constant changed and the codelet rebuilt and re-verified, even though nothing on the +controller side needs recompiling. Lowering it is fine — a 2x2 config just uses less of the +row. + +This does not fail silently. The controller validates its configured geometry against what +the RAN reports in the first slot and refuses to publish on a mismatch, and the gNB-side +publish helper refuses a slot larger than the row rather than writing a prefix. + +### No verification at codelet load time + +Codelets are verified **offline**, at build time (see [Codelets](#codelets)). The gNB does +not re-verify on load: jbpf JIT-compiles with ubpf, which emits no memory bounds checks. So +an object that bypasses `make` — hand-copied onto a pod, say — is loaded unchecked. + +Mitigations in place: `make` refuses to replace a codelet object that does not verify, and +the gNB-side publish helper range-checks its source pointer at runtime against a window the +hook publishes, independently of whether the codelet was verified. A load-time +`jbpf_verify()` call before the LCM request is the remaining gap. diff --git a/apply_jbpf_patches.sh b/apply_jbpf_patches.sh new file mode 100755 index 0000000..bdbe067 --- /dev/null +++ b/apply_jbpf_patches.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +# +# Apply this repo's patches against the jbpf submodule. +# +# Run after `git submodule update --init --recursive` and before configuring. +# build.sh calls it as step 2; it is also safe to run by hand. +# +# Idempotent: an already-patched tree is detected and skipped. +set -euo pipefail + +REPO_DIR="$(cd "$(dirname "$0")" && pwd)" +JBPF_DIR="${REPO_DIR}/jbpf" +OURS_DIR="${REPO_DIR}/jbpf_patches" + +if [ ! -f "${JBPF_DIR}/CMakeLists.txt" ]; then + echo "ERROR: jbpf submodule at ${JBPF_DIR} is not populated." >&2 + echo " Run: git submodule update --init --recursive" >&2 + exit 1 +fi + +# run_patch +# +# One call site for both back-ends so the two never drift in what they +# consider "applies". Returns the tool's exit status; prints nothing. +run_patch() { + local mode="$1" tree="$2" file="$3" dir="$4" kind="$5" + local -a args=() + if [ "${dir}" = reverse ]; then args+=(--reverse); fi + + # ${args[@]+...}: a forward+real call leaves args empty, and bash 3.2 (still + # the /bin/bash on macOS) treats "${args[@]}" as unbound under `set -u`. + if [ "${mode}" = git ]; then + if [ "${kind}" = dry ]; then args+=(--check); fi + git -C "${tree}" apply ${args[@]+"${args[@]}"} "${file}" >/dev/null 2>&1 + else + if [ "${kind}" = dry ]; then args+=(--dry-run); fi + ( cd "${tree}" && patch -p1 --silent --force --no-backup-if-mismatch \ + ${args[@]+"${args[@]}"} <"${file}" ) >/dev/null 2>&1 + fi +} + +apply_patch() { + local patch="$1" + local patch_file="${OURS_DIR}/${patch}" + + if [ ! -f "${patch_file}" ]; then + echo "ERROR: ${patch} is missing from ${OURS_DIR}." >&2 + exit 1 + fi + + # git apply where the tree is a real checkout, patch(1) where it is not. + # The fallback is not hypothetical: deployment rsyncs this tree into the pod + # without .git, and there `git apply` fails at repository DISCOVERY -- + # indistinguishable, from the exit status alone, from a conflicting patch. + local mode=git + git -C "${JBPF_DIR}" rev-parse --git-dir >/dev/null 2>&1 || mode=patch + + if run_patch "${mode}" "${JBPF_DIR}" "${patch_file}" reverse dry; then + echo "==> jbpf: ${patch} already applied, skipping" + elif run_patch "${mode}" "${JBPF_DIR}" "${patch_file}" forward dry; then + echo "==> jbpf: applying ${patch}" + run_patch "${mode}" "${JBPF_DIR}" "${patch_file}" forward real + else + echo "ERROR: ${patch} does not apply cleanly to jbpf (nor is it already" >&2 + echo " applied). The tree may have local modifications; inspect it," >&2 + echo " or re-init with:" >&2 + echo " git submodule update --init --recursive --force jbpf" >&2 + exit 1 + fi +} + +# Upstream jbpf_time_get_ns() stamps CLOCK_REALTIME, and that is the clock the +# slot codelet writes into codelet_ts_ns. The gNB, this controller and the dApp +# all read CLOCK_MONOTONIC, so leaving it unpatched puts codelet_ts_ns in a +# different epoch from gnb_ts_ns: gnb_to_codelet_us then comes out silently +# wrong rather than failing. +apply_patch jbpf_monotonic_time.patch + +echo "==> All E3Controller jbpf patches applied." diff --git a/bench/bench_stage_recording.cpp b/bench/bench_stage_recording.cpp new file mode 100644 index 0000000..99bf1ca --- /dev/null +++ b/bench/bench_stage_recording.cpp @@ -0,0 +1,390 @@ +/* + * What does recording a slot cost? + * + * This exists because the mechanism it replaces got the answer wrong. The + * per-slot stage CSV that used to live at the tail of E3SMLayer1::on_sample + * opened an ofstream, wrote a row and *flushed* it, once per slot, on the + * pipeline worker thread - so every number it reported included the cost of + * reporting it. Replacing it is only justified if the replacement is cheap, and + * "cheap" has to be measured rather than asserted. + * + * Three arms, one binary, no slot work: + * + * none the loop and the argument marshalling, nothing else - the floor + * latrec exactly what on_sample now emits per slot, from the same header + * csv the removed writer, reproduced field for field + * + * Deliberately no "real slot work" arm. Under the default `shm.writer: gnb` the + * gNB converts the grid and writes the shared-memory row itself, so the + * controller does no data movement at all - a slot-work arm here would measure + * something this process does not do. And against real work the stamps were + * never resolvable anyway: they are a fraction of a percent of it, so the only + * honest output was a bound. The interesting number lives here, where a + * per-slot write(2) and a handful of stores differ by an order of magnitude and + * separate cleanly. + * + * Method notes that matter for believing the output: + * - Batch timing, not per-iteration. One clock_gettime costs more than a + * latrec stamp, so timing each iteration would measure the timing. + * - Arms are interleaved and rotated across batches, so a runner that slows + * down partway through penalises every arm equally instead of whichever one + * ran last. + * - Median across batches, with min/max, rather than a mean: one descheduled + * batch should not move the answer. + * - The clock's own cost is measured and reported, since it is the floor + * under every latrec stamp. + * + * The `csv` arm is a benchmark reference, not a recording facility: nothing in + * the controller writes a per-slot CSV any more. It is kept so the mechanism + * can still be priced after it is gone. + * + * Build with a libe3 configured -DLIBE3_ENABLE_LATREC=ON, or the latrec arm + * measures the no-op stubs - which is itself worth checking, see --mode + * compiled-out. + */ +#include "e3sm/l1_kpm/l1_kpm_trace.h" + +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace trace = e3sm_l1kpm_trace; + +namespace { + +using clock_type = std::chrono::steady_clock; + +/* One slot's worth of plausible stage inputs. Values are arbitrary but + * realistic in magnitude, so the CSV arm formats the same number of digits it + * would in production (integer formatting cost scales with digit count). */ +struct SlotFacts { + uint64_t seq; + uint32_t sfn; + uint16_t abs_slot; + /* Absolute times fed to the back-dated stamps. Offsets from `now` are tiny + * on purpose - see slot_facts(). */ + uint64_t gnb_ts_ns; + uint64_t codelet_entry_ts_ns; + uint64_t codelet_ts_ns; + uint64_t dispatch_ts_ns; + uint32_t bytes_written; + uint32_t nof_subc; + std::size_t encoded_bytes; + std::size_t subscribers; +}; + +SlotFacts slot_facts(uint64_t i) +{ + /* The two A1 stamps are back-dated, so they have to ascend across + * iterations or the ring's monotone floor kicks in and we would be timing + * the clamp path instead of the normal one. + * + * In production that is free: slots are >=500 us apart and A1 is ~70 us + * deep, so slot N+1's gnb stamp is comfortably later than slot N's last + * record. This loop iterates ~130 ns apart, so a 70 us back-date could not + * possibly ascend. The offsets below are therefore nanoseconds, not + * microseconds - and that costs nothing in fidelity, because what a stamp + * costs does not depend on the magnitude of the timestamp it carries. It + * depends on the code path, and this is the same one. + * + * The CSV arm does not use these at all; it formats fixed realistic + * durations, so its integer-formatting cost stays representative. */ + const uint64_t now = latrec_tnow(); + SlotFacts f{}; + f.seq = i + 1; + f.sfn = static_cast(i % 1024); + f.abs_slot = static_cast(i % 20); + f.gnb_ts_ns = now - 4; + f.codelet_entry_ts_ns = now - 3; + f.codelet_ts_ns = now - 2; + f.dispatch_ts_ns = now - 1; + f.bytes_written = 733824; + f.nof_subc = 3276; + f.encoded_bytes = 96; + f.subscribers = 1; + return f; +} + +/* ---------------- the three ways of recording a slot ---------------- */ + +/* Arm `none`: the loop and the argument marshalling, nothing else. This is the + * floor the other two are measured against, not zero. */ +void record_none(const SlotFacts& f) +{ + asm volatile("" : : "r"(&f) : "memory"); +} + +/* Arm `latrec`: exactly what E3SMLayer1::on_sample emits per slot - the same + * inline functions, from the same header, in the same order. */ +void record_latrec(const SlotFacts& f) +{ + trace::record_begin(f.seq, f.sfn, f.abs_slot, f.gnb_ts_ns, f.codelet_entry_ts_ns); + trace::process_begin(f.seq, f.codelet_ts_ns, f.dispatch_ts_ns, f.bytes_written); + trace::encode_begin(f.seq, f.bytes_written); + trace::encode_done(f.seq, f.encoded_bytes); + trace::bind_libe3(f.seq); + trace::emit_tail(f.seq, f.subscribers); +} + +/* Arm `csv`: the removed mechanism, reproduced field for field - including the + * saturating subtraction it used, and the flush that is the whole point. */ +class CsvArm { +public: + explicit CsvArm(const std::string& path) : out_(path, std::ios::out | std::ios::trunc) + { + out_ << "slot_seq," + "gnb_to_codelet_us,codelet_publish_us," + "codelet_to_dispatch_us,dispatch_to_handler_us," + "shm_ns,encode_ns,emit_ns,nof_subc,iq_bytes\n"; + } + + bool ok() const { return out_.is_open(); } + + void record(const SlotFacts& f) + { + /* Fixed, realistic stage durations - the measured shape at 30 kHz SCS: + * ~5 us jbpf dispatch, ~66 us convert + row write, then the ring + * transit and the queue wait. Literals rather than differences of + * f's timestamps, because those are nanoseconds apart here (see + * slot_facts) and would format far fewer digits than production. */ + out_ << f.seq << ',' + << 5 << ',' + << 66 << ',' + << 9 << ',' + << 15 << ',' + << 0 << ',' + << 1850 << ',' + << 2400 << ',' + << f.nof_subc << ',' + << f.bytes_written << '\n'; + out_.flush(); // the whole point: one flush per slot, on the data path + } + +private: + std::ofstream out_; +}; + +/* ---------------- statistics ---------------- */ + +struct Summary { + double median_ns; + double min_ns; + double max_ns; +}; + +Summary summarise(std::vector batch_ns_per_iter) +{ + std::sort(batch_ns_per_iter.begin(), batch_ns_per_iter.end()); + Summary s{}; + const std::size_t n = batch_ns_per_iter.size(); + s.median_ns = (n % 2) ? batch_ns_per_iter[n / 2] + : 0.5 * (batch_ns_per_iter[n / 2 - 1] + batch_ns_per_iter[n / 2]); + s.min_ns = batch_ns_per_iter.front(); + s.max_ns = batch_ns_per_iter.back(); + return s; +} + +template +double time_batch_ns_per_iter(Fn&& fn, uint64_t iters, uint64_t& counter) +{ + const auto t0 = clock_type::now(); + for (uint64_t i = 0; i < iters; ++i) { + fn(counter++); + } + const auto t1 = clock_type::now(); + const double total = + static_cast(std::chrono::duration_cast(t1 - t0).count()); + return total / static_cast(iters); +} + +/* Percentage of one slot's budget at 30 kHz SCS. The number that actually + * answers "was the old mechanism a problem?". */ +double pct_of_slot_budget(double ns) { return 100.0 * ns / 500000.0; } + +void print_row(const char* name, const Summary& s) +{ + std::printf("| %-22s | %12.1f | %10.1f | %10.1f | %8.4f%% |\n", + name, s.median_ns, s.min_ns, s.max_ns, pct_of_slot_budget(s.median_ns)); +} + +bool latrec_compiled_in() +{ +#ifdef LIBE3_ENABLE_LATREC + return true; +#else + return false; +#endif +} + +int run_instrumentation(uint64_t iters, uint64_t batches, const std::string& csv_path) +{ + CsvArm csv(csv_path); + if (!csv.ok()) { + std::fprintf(stderr, "cannot open %s for the csv arm\n", csv_path.c_str()); + return 1; + } + + std::vector none_b, latrec_b, csv_b; + uint64_t counter = 0; + + /* Warm up every arm before measuring anything: first-touch page faults on + * the ring mapping and the ofstream buffer would otherwise be charged to + * whichever arm ran first. */ + for (uint64_t i = 0; i < 512; ++i) { + const SlotFacts f = slot_facts(counter++); + record_none(f); + record_latrec(f); + csv.record(f); + } + + for (uint64_t b = 0; b < batches; ++b) { + /* Rotate the order so no arm is permanently first or last. */ + const int order[3] = {static_cast(b % 3), + static_cast((b + 1) % 3), + static_cast((b + 2) % 3)}; + for (int slot = 0; slot < 3; ++slot) { + switch (order[slot]) { + case 0: + none_b.push_back(time_batch_ns_per_iter( + [](uint64_t i) { record_none(slot_facts(i)); }, iters, counter)); + break; + case 1: + latrec_b.push_back(time_batch_ns_per_iter( + [](uint64_t i) { record_latrec(slot_facts(i)); }, iters, counter)); + break; + default: + csv_b.push_back(time_batch_ns_per_iter( + [&csv](uint64_t i) { csv.record(slot_facts(i)); }, iters, counter)); + break; + } + } + } + + std::printf("\n### Recording one slot\n\n"); + std::printf("| %-22s | %12s | %10s | %10s | %9s |\n", + "arm", "median ns", "min ns", "max ns", "of slot"); + std::printf("|------------------------|--------------|------------|------------|-----------|\n"); + const Summary sn = summarise(none_b); + const Summary sl = summarise(latrec_b); + const Summary sc = summarise(csv_b); + print_row("none (loop floor)", sn); + print_row(latrec_compiled_in() ? "latrec (5 records)" : "latrec (STUBS - off)", sl); + print_row("csv (removed)", sc); + + const double latrec_net = sl.median_ns - sn.median_ns; + const double csv_net = sc.median_ns - sn.median_ns; + std::printf("\nNet of the loop floor: latrec %.1f ns/slot, csv %.1f ns/slot.\n", + latrec_net, csv_net); + if (latrec_net > 0.0) { + std::printf("The removed CSV cost %.0fx what the stamps cost.\n", csv_net / latrec_net); + } + std::printf("Clock read (the floor under every stamp): %u ns.\n", + latrec_measure_clock_ns()); + + /* A clamp here would mean the synthetic times did not ascend, i.e. the + * bench is exercising the wrong path and its latrec arm is not comparable + * with production. Report it rather than letting it pass silently. */ + if (const uint64_t clamped = trace::clamped(); clamped != 0) { + std::printf("\nWARNING: %llu stamp(s) were clamped. The synthetic slot times are not\n" + " ascending, so the latrec arm above is not measuring the normal\n" + " path. Treat the number as invalid.\n", + static_cast(clamped)); + return 1; + } + if (!latrec_compiled_in()) { + std::printf("\nNOTE: built without LIBE3_ENABLE_LATREC, so the latrec arm above is\n" + " measuring the no-op stubs, not the recorder.\n"); + } + return 0; +} + +int run_compiled_out() +{ + /* A timing test cannot prove absence - only a build can. This reports what + * the binary was built with so CI can assert on it; the companion check is + * `nm` over a latrec-OFF build finding no global latrec symbols. */ + std::printf("LIBE3_ENABLE_LATREC=%s\n", latrec_compiled_in() ? "ON" : "OFF"); +#ifdef LATREC_DEFAULT_DIR + std::printf("LATREC_DEFAULT_DIR=%s\n", LATREC_DEFAULT_DIR); +#else + std::printf("LATREC_DEFAULT_DIR=(unset)\n"); +#endif + return 0; +} + +void usage(const char* prog) +{ + std::fprintf(stderr, + "Usage: %s [--mode instrumentation|compiled-out|all]\n" + " [--iters N] [--batches K] [--csv-path P] [--latrec-dir D]\n\n" + " --mode which measurement to run (default: all)\n" + " --iters iterations per batch (default: 2000)\n" + " --batches batches per arm (default: 15)\n" + " --csv-path scratch file for the csv arm (default: ./bench_csv_arm.log)\n" + " --latrec-dir where to write stage-record rings\n", prog); +} + +} // namespace + +int main(int argc, char** argv) +{ + std::string mode = "all"; + std::string csv_path = "./bench_csv_arm.log"; + std::string latrec_dir; + uint64_t iters = 2000; + uint64_t batches = 15; + + enum { OPT_MODE = 1000, OPT_ITERS, OPT_BATCHES, OPT_CSV_PATH, OPT_LATREC_DIR }; + static struct option opts[] = { + {"mode", required_argument, nullptr, OPT_MODE}, + {"iters", required_argument, nullptr, OPT_ITERS}, + {"batches", required_argument, nullptr, OPT_BATCHES}, + {"csv-path", required_argument, nullptr, OPT_CSV_PATH}, + {"latrec-dir", required_argument, nullptr, OPT_LATREC_DIR}, + {"help", no_argument, nullptr, 'h'}, + {nullptr, 0, nullptr, 0} + }; + + int o; + while ((o = getopt_long(argc, argv, "h", opts, nullptr)) != -1) { + switch (o) { + case OPT_MODE: mode = optarg; break; + case OPT_ITERS: iters = std::strtoull(optarg, nullptr, 10); break; + case OPT_BATCHES: batches = std::strtoull(optarg, nullptr, 10); break; + case OPT_CSV_PATH: csv_path = optarg; break; + case OPT_LATREC_DIR: latrec_dir = optarg; break; + default: usage(argv[0]); return o == 'h' ? 0 : 2; + } + } + if (mode != "all" && mode != "instrumentation" && mode != "compiled-out") { + usage(argv[0]); + return 2; + } + + if (!latrec_dir.empty()) { + latrec_set_output_dir(latrec_dir.c_str()); + } + /* Single-threaded, so one ring covers every arm. Opened here, before any + * measurement, so the mapping is faulted in off the measured path. */ + trace::open_ring(); + + std::printf("## Stage-recording cost\n"); + std::printf("\nlatrec: %s. Batches: %llu, iterations/batch: %llu.\n", + latrec_compiled_in() ? "compiled in" : "NOT compiled in", + static_cast(batches), + static_cast(iters)); + + int rc = 0; + if (mode == "all" || mode == "compiled-out") rc |= run_compiled_out(); + if (mode == "all" || mode == "instrumentation") rc |= run_instrumentation(iters, batches, csv_path); + return rc; +} diff --git a/build.sh b/build.sh index 63f3cf2..1b1af8e 100755 --- a/build.sh +++ b/build.sh @@ -12,26 +12,38 @@ # grammar expects is produced. We also apt-install a small set of extras # the E3Controller itself needs (python + pip for jbpf's build). # 1. fetch submodules (libe3 @ pinned tag, jbpf) -# 2. init + patch jbpf's own 3p submodules +# 2. init + patch jbpf's own 3p submodules, then apply our patch to jbpf core # 3. configure libe3, stage asn1c's BOOLEAN.* skeletons (toolchain workaround), # then build + INSTALL to /usr/local (both encoders -> runtime --encoding) # 4. build the E3Controller (jbpf is built in-tree via add_subdirectory) # -# Usage: ./build.sh [--install-deps] +# Usage: ./build.sh [--install-deps] [--latrec] # --install-deps Install all system packages before building. Debian/Ubuntu # only (uses apt-get). Uses sudo if not already root. +# --latrec Build libe3 with -DLIBE3_ENABLE_LATREC=ON, so the stage +# recorder is compiled in. Off by default: a normal build has +# no recorder in the process and the controller's own stamps +# compile to nothing. LIBE3_ENABLE_LATREC is a PUBLIC compile +# definition on libe3::libe3, so this one flag reaches the +# controller too -- there is nothing to pass twice, and a +# mismatch is a link error rather than a silently untraced +# build. Set LATREC_DEFAULT_DIR= alongside it to move +# the compiled-in default ring directory off /tmp/latrec. # # Override parallelism with JOBS= ./build.sh. +# Extra configure flags for the controller: E3C_CMAKE_ARGS='-DUSE_NATIVE=OFF'. set -euo pipefail cd "$(dirname "$0")" INSTALL_DEPS=0 +ENABLE_LATREC=0 for arg in "$@"; do case "$arg" in --install-deps|-d) INSTALL_DEPS=1 ;; - -h|--help) sed -n '2,20p' "$0"; exit 0 ;; + --latrec) ENABLE_LATREC=1 ;; + -h|--help) sed -n '2,34p' "$0"; exit 0 ;; *) echo "ERROR: unknown argument '$arg'" >&2 - echo "Usage: $0 [--install-deps]" >&2 + echo "Usage: $0 [--install-deps] [--latrec]" >&2 exit 2 ;; esac done @@ -126,41 +138,64 @@ echo "==> [1/4] Fetching submodules (libe3, jbpf)" git submodule update --init --recursive # --------------------------------------------------------------------------- -# Step 2: jbpf's own 3p submodules +# Step 2: jbpf's own 3p submodules, then jbpf core # --------------------------------------------------------------------------- echo "==> [2/4] Initialising + patching jbpf 3p submodules" ( cd jbpf && bash ./init_and_patch_submodules.sh ) +# Then our own patches against jbpf core, which that script knows nothing +# about (it is upstream jbpf's, so edits to it are lost on re-clone). +bash ./apply_jbpf_patches.sh + # --------------------------------------------------------------------------- # Step 3: libe3 # --------------------------------------------------------------------------- echo "==> [3/4] Building + installing libe3 (ASN.1 + JSON) to /usr/local" # JSON encoding needs nlohmann_json >= 3.11 on the system (libe3's floor); install # it (header-only) or let libe3's FetchContent fetch it when the host has internet. -cmake -S libe3 -B libe3/build -DLIBE3_ENABLE_ASN1=ON -DLIBE3_ENABLE_JSON=ON \ - -DLIBE3_BUILD_EXAMPLES=OFF -DLIBE3_BUILD_TESTS=OFF - -# --- toolchain shim: supply BOOLEAN.* to libe3's E3AP runtime -------------- -# libe3 0.0.4's messages/asn1/V1/e3ap-1.0.0.cmake hard-lists BOOLEAN.{c,h} and -# BOOLEAN_{aper,print,rfill,uper,xer}.c as asn1c outputs, but its E3AP grammar -# never uses BOOLEAN, so the mouse07410 asn1c fork does NOT emit them and the -# build fails on the missing sources. Rather than patch libe3, drop asn1c's own -# BOOLEAN skeletons into libe3's generated dir before the build (asn1c won't -# overwrite them). libe3 then owns asn_DEF_BOOLEAN; E3Controller links it -# instead of compiling its own (see src/e3sm/asn/CMakeLists.txt). Re-run this -# script after a clean (rm -rf libe3/build) so the copy is re-staged. -BOOLEAN_FILES="BOOLEAN.c BOOLEAN.h BOOLEAN_aper.c BOOLEAN_print.c BOOLEAN_rfill.c BOOLEAN_uper.c BOOLEAN_xer.c" -SKEL="${ASN1C_SKELETON_DIR:-}" -if [ -z "${SKEL}" ]; then - for d in /opt/asn1c/share/asn1c /usr/local/share/asn1c /usr/share/asn1c; do - [ -f "$d/BOOLEAN.c" ] && { SKEL="$d"; break; } - done +# +# CMAKE_BUILD_TYPE is NOT optional. libe3's own CMakeLists never defaults it, and +# this line used to omit it, so libe3 compiled with NO -O flag at all: +# +# CXX_FLAGS = -fPIC -Wall -Wextra ... -std=c++17 # and nothing else +# +# That is the E3AP encoder, the ZMQ connector and the outbound lock-free queue -- +# i.e. exactly the stages the RAN->dApp latency budget attributes its residual to +# ("queueing within libe3, E3AP encoding, and the transmission"). An unoptimised +# build inflates all three and the only symptom is a slower number, which is +# indistinguishable from the system genuinely being slow. +# +# Overridable for a debug build: LIBE3_BUILD_TYPE=RelWithDebInfo ./build.sh +LIBE3_BUILD_TYPE="${LIBE3_BUILD_TYPE:-Release}" +echo " libe3 CMAKE_BUILD_TYPE=${LIBE3_BUILD_TYPE}" +LIBE3_CMAKE_ARGS=( + -DCMAKE_BUILD_TYPE="${LIBE3_BUILD_TYPE}" + -DLIBE3_ENABLE_ASN1=ON + -DLIBE3_ENABLE_JSON=ON + -DLIBE3_BUILD_EXAMPLES=OFF + -DLIBE3_BUILD_TESTS=OFF +) +if [ "$ENABLE_LATREC" -eq 1 ]; then + echo " latrec: ON (stage recorder compiled in)" + LIBE3_CMAKE_ARGS+=( -DLIBE3_ENABLE_LATREC=ON ) + # libe3 only defaults LATREC_DEFAULT_DIR into its build tree when it is + # building its own tests, which we turn off -- so without this the compiled-in + # default stays latrec.h's /tmp/latrec. Pass it through when the caller names + # one; logging.latrec_dir in the YAML overrides it per run either way. + if [ -n "${LATREC_DEFAULT_DIR:-}" ]; then + echo " latrec: default ring directory ${LATREC_DEFAULT_DIR}" + LIBE3_CMAKE_ARGS+=( "-DLATREC_DEFAULT_DIR=${LATREC_DEFAULT_DIR}" ) + fi fi -[ -n "${SKEL}" ] || { echo "ERROR: asn1c BOOLEAN skeletons not found; set ASN1C_SKELETON_DIR=

"; exit 1; } -echo " supplying BOOLEAN skeletons from ${SKEL} -> libe3/build/messages" -mkdir -p libe3/build/messages -for f in ${BOOLEAN_FILES}; do cp "${SKEL}/${f}" libe3/build/messages/; done -# --------------------------------------------------------------------------- +cmake -S libe3 -B libe3/build "${LIBE3_CMAKE_ARGS[@]}" + +# No BOOLEAN.* skeleton staging here. It existed because Spectrum-ConfigControl +# used BOOLEAN while libe3's E3AP grammar did not, so asn1c never emitted the +# skeleton on libe3's side and we staged it there to have libe3 compile it for +# us. Spectrum-ConfigControl is no longer compiled (src/e3sm/asn/CMakeLists.txt) +# and nothing references asn_DEF_BOOLEAN, so the staging had nothing left to +# supply -- while still aborting the build outright on any host without asn1c's +# reference skeletons on disk. cmake --build libe3/build -j"${JOBS}" $SUDO cmake --install libe3/build @@ -169,7 +204,12 @@ $SUDO cmake --install libe3/build # Step 4: E3Controller # --------------------------------------------------------------------------- echo "==> [4/4] Building E3Controller" -cmake -S . -B build -DINITIALIZE_SUBMODULES=OFF -cmake --build build -j"${JOBS}" --target e3_controller +# E3C_CMAKE_ARGS lets a caller add configure flags without editing this script. +# CI sets -DUSE_NATIVE=OFF: -march=native is right on a deployment host but wrong +# for a measurement run on whatever CPU a shared runner happens to be, since it +# makes numbers incomparable between runs. +# shellcheck disable=SC2086 +cmake -S . -B build -DINITIALIZE_SUBMODULES=OFF ${E3C_CMAKE_ARGS:-} +cmake --build build -j"${JOBS}" --target e3_controller bench_stage_recording echo "==> Done: $(pwd)/out/bin/e3_controller" diff --git a/cmake/libe3_stamp.h.in b/cmake/libe3_stamp.h.in new file mode 100644 index 0000000..1709dc7 --- /dev/null +++ b/cmake/libe3_stamp.h.in @@ -0,0 +1,14 @@ +/* Generated by CMake -- do not edit. Records which installed libe3 this binary + * was LINKED AGAINST, so a stale snapshot is detectable at runtime. + * + * The controller links libe3 STATICALLY (libe3Targets.cmake declares + * libe3::libe3 as STATIC IMPORTED -> liblibe3.a), so reinstalling libe3 does + * NOT affect an already-built e3_controller. Nothing about that is visible at + * runtime -- a changed SLEEP_DURATION or encoder simply is not there -- which is + * why this stamp exists. */ +#pragma once + +#define LIBE3_STAMP_VERSION "@LIBE3_STAMP_VERSION@" +#define LIBE3_STAMP_LIB "@LIBE3_STAMP_LIB@" +#define LIBE3_STAMP_SLEEP_US "@LIBE3_STAMP_SLEEP_US@" +#define LIBE3_STAMP_BUILDTYPE "@LIBE3_STAMP_BUILDTYPE@" diff --git a/codelets/Makefile b/codelets/Makefile new file mode 100644 index 0000000..ec5db54 --- /dev/null +++ b/codelets/Makefile @@ -0,0 +1,63 @@ +# codelets/Makefile — build and verify every codelet in this directory. +# +# make build + verify all codelets +# make verify re-verify existing objects +# make import-contract copy ocudu's ctx header into include/ (for builds +# without an ocudu checkout, e.g. a container) +# make check-contract fail if the imported copy has drifted from ocudu +# make clean + +include Makefile.defs + +.PHONY: all verify clean cleanall import-contract check-contract $(CODELET_DIRS) + +all: check-env check-contract $(CODELET_DIRS) + +$(CODELET_DIRS): + $(MAKE) -C $@ + +verify: + @for d in $(CODELET_DIRS); do $(MAKE) -C $$d verify || exit 1; done + +clean cleanall: + @for d in $(CODELET_DIRS); do $(MAKE) -C $$d $@ || exit 1; done + +# ---- ctx contract (A2a) ---- +# +# ocudu is the source of truth: include/ocudu/janus/jbpf_srsran_contexts.h. +# `import-contract` takes a copy for environments where that tree is absent; +# `check-contract` is a no-op there and a hard diff where it is present, so +# drift surfaces as a build failure instead of an ABI mismatch at runtime. + +CONTRACT_HDR := jbpf_srsran_contexts.h + +import-contract: + @if [ ! -f "$(OCUDU_JANUS_INC)/$(CONTRACT_HDR)" ]; then \ + echo "ERROR: $(OCUDU_JANUS_INC)/$(CONTRACT_HDR) not found." >&2; \ + echo " Pass OCUDU_DIR=/path/to/ocudu-e3" >&2; \ + exit 1; \ + fi + @mkdir -p $(CONTRACT_INC) + cp "$(OCUDU_JANUS_INC)/$(CONTRACT_HDR)" "$(CONTRACT_INC)/$(CONTRACT_HDR)" + @echo "imported $(CONTRACT_HDR) from $(OCUDU_JANUS_INC)" + +check-contract: + @if [ ! -f "$(OCUDU_JANUS_INC)/$(CONTRACT_HDR)" ]; then \ + if [ -f "$(CONTRACT_INC)/$(CONTRACT_HDR)" ]; then \ + echo "note: no ocudu tree at $(OCUDU_DIR); using imported $(CONTRACT_HDR)"; \ + else \ + echo "ERROR: no ctx contract available." >&2; \ + echo " Either pass OCUDU_DIR=/path/to/ocudu-e3," >&2; \ + echo " or run 'make import-contract' where that tree exists." >&2; \ + exit 1; \ + fi; \ + elif [ -f "$(CONTRACT_INC)/$(CONTRACT_HDR)" ] && \ + ! diff -q "$(OCUDU_JANUS_INC)/$(CONTRACT_HDR)" "$(CONTRACT_INC)/$(CONTRACT_HDR)" >/dev/null; then \ + echo "ERROR: ctx contract drift." >&2; \ + echo " ocudu: $(OCUDU_JANUS_INC)/$(CONTRACT_HDR)" >&2; \ + echo " imported: $(CONTRACT_INC)/$(CONTRACT_HDR)" >&2; \ + diff -u "$(CONTRACT_INC)/$(CONTRACT_HDR)" "$(OCUDU_JANUS_INC)/$(CONTRACT_HDR)" >&2 || true; \ + echo "Run 'make import-contract' to resync, then rebuild ALL codelets:" >&2; \ + echo "a changed ctx layout shifts every offset they read through." >&2; \ + exit 1; \ + fi diff --git a/codelets/Makefile.common b/codelets/Makefile.common new file mode 100644 index 0000000..10e01c7 --- /dev/null +++ b/codelets/Makefile.common @@ -0,0 +1,66 @@ +# codelets/Makefile.common — build + verify rules for one codelet directory. +# +# Included by each codelet's Makefile after Makefile.defs. + +.PHONY: all verify clean cleanall check-cc check-verifier + +all: check-cc check-env check-verifier $(C_OBJECTS) + +# Fail with an explanation rather than GCC's bare +# "unrecognized command-line option '-target'". +check-cc: + @$(CC) -target bpf --version >/dev/null 2>&1 || { \ + echo "ERROR: '$(CC)' cannot target eBPF." >&2; \ + echo " Codelets need clang; GCC rejects -target." >&2; \ + echo " Install clang, or point CC at it: make CC=clang-14" >&2; \ + exit 1; \ + } + +# Compile, then VERIFY — and let verification failure fail the build. +# +# The rule this replaces (jrtc-apps/codelets/Makefile.common) read: +# +# - $(VERIFIER_BIN) $@ || echo "$<: Failed verification" +# +# which swallowed failure twice over: the leading `-` tells make to ignore the +# exit status, and `|| echo` masks it again. That is the mechanism by which +# these codelets have never actually had to verify. Since the gNB does no +# verification at load time (ubpf only — see codelets/verifier/e3_verifier_cli.cpp +# for the details), this build step is the only safety gate that ever runs on +# them, so it has to be able to fail. +# +# No section argument: each object holds a single program, and the verifier +# resolves the program type from the ELF section prefix. +# Build via a temporary and only move into place once verification passes. +# Two reasons this matters: +# 1. The .o files are TRACKED and are what gets deployed to the pod, so a +# failed compile must not destroy the last good one. clang removes its +# output file on error, which is exactly what happens otherwise. +# 2. It makes it impossible for an unverified object to replace a verified +# one, even transiently. +$(C_OBJECTS): %.o: %.c + @echo "--- compiling $< ---" + $(CC) $(CFLAGS) $(INC) -c $< -o $@.tmp + @echo "--- verifying $< ---" + @$(E3_VERIFIER) $@.tmp || { rm -f $@.tmp; echo "NOT replacing $@ — previous object left in place" >&2; exit 1; } + @mv $@.tmp $@ + +# Checked here rather than as a prerequisite of the objects: a file target with +# a recipe is force-remade by `make -B`, which made a perfectly good verifier +# look missing. +check-verifier: + @if [ ! -x "$(E3_VERIFIER)" ]; then \ + echo "ERROR: verifier not found (or not executable) at $(E3_VERIFIER)" >&2; \ + echo " Build it first: cd $(E3_ROOT) && ./build.sh" >&2; \ + echo " (or pass E3_VERIFIER=/path/to/e3_verifier_cli)" >&2; \ + exit 1; \ + fi + +# Verify already-built objects without recompiling. +verify: check-verifier $(C_OBJECTS) + @for o in $(C_OBJECTS); do $(E3_VERIFIER) $$o || exit 1; done + +clean: + rm -f *.o.tmp + +cleanall: clean diff --git a/codelets/Makefile.defs b/codelets/Makefile.defs new file mode 100644 index 0000000..0528520 --- /dev/null +++ b/codelets/Makefile.defs @@ -0,0 +1,155 @@ +# codelets/Makefile.defs — toolchain and include paths for E3Controller codelets. +# +# Self-contained: NO dependency on jrtc-apps. Previously these rules lived in +# jrtc-apps/codelets/, most of which was protobuf/jrtc machinery (PROTO_AND_SCHEMA, +# jbpf_protobuf_cli serde, ctypesgen, USE_JRTC) that this project never used — +# our codelets publish through jbpf output maps and the controller's own +# ASN.1/JSON encoders. What remains is the ten lines we actually need. +# +# Overridable on the command line, e.g.: +# make OCUDU_DIR=/src/ocudu-e3 + +# `all` must stay the default goal. This file is included FIRST by every codelet +# Makefile, so without this the first target defined below (show-config) would +# become the default and a bare `make` would print diagnostics instead of +# building. +.DEFAULT_GOAL := all + +CODELETS_DIR := $(patsubst %/,%,$(dir $(abspath $(lastword $(MAKEFILE_LIST))))) +E3_ROOT := $(patsubst %/,%,$(dir $(CODELETS_DIR))) + +# ---- jbpf (the E3Controller submodule; built by build.sh into $(E3_ROOT)/out) ---- +JBPF_DIR ?= $(E3_ROOT)/jbpf +JBPF_OUT_DIR ?= $(E3_ROOT)/out + +# ---- ocudu: source of truth for the hook context contract ---- +# The ctx structs are defined by the hook PRODUCER, so ocudu owns them. We put +# its headers on the include path rather than keeping a copy here. Three +# byte-identical copies of jbpf_srsran_contexts.h used to float around the +# workspace with nothing enforcing agreement; a field added on the RAN side +# would silently shift every offset the codelet reads through. +OCUDU_DIR ?= $(E3_ROOT)/../ocudu-e3 +OCUDU_JANUS_INC ?= $(OCUDU_DIR)/include/ocudu/janus +OCUDU_SPECS_INC ?= $(OCUDU_DIR)/ocudu_janus/verifier/specs + +# Local import target for environments without an ocudu checkout (e.g. a build +# container). `make import-contract` populates it; `make check-contract` fails +# on drift. OCUDU_JANUS_INC comes first on the include path so the live tree +# always wins when it is present. +CONTRACT_INC := $(CODELETS_DIR)/include + +# ---- offline verifier (built by cmake from codelets/verifier/) ---- +E3_VERIFIER ?= $(JBPF_OUT_DIR)/bin/e3_verifier_cli + +# Absolutise every path that can be overridden. The top-level Makefile recurses +# with `$(MAKE) -C `, so a relative override like +# `make OCUDU_DIR=../ocudu-e3` would otherwise be re-resolved against each +# subdirectory's CWD and silently fail to find the contract. +override JBPF_DIR := $(abspath $(JBPF_DIR)) +override JBPF_OUT_DIR := $(abspath $(JBPF_OUT_DIR)) +override OCUDU_DIR := $(abspath $(OCUDU_DIR)) +override OCUDU_JANUS_INC := $(abspath $(OCUDU_JANUS_INC)) +override OCUDU_SPECS_INC := $(abspath $(OCUDU_SPECS_INC)) +override E3_VERIFIER := $(abspath $(E3_VERIFIER)) + +# Pass them down explicitly so sub-makes get the resolved absolute values. +export JBPF_DIR JBPF_OUT_DIR OCUDU_DIR OCUDU_JANUS_INC OCUDU_SPECS_INC E3_VERIFIER + +# ---- toolchain ---- +# +# Codelets are eBPF objects, so the compiler MUST be clang: `-target bpf` is a +# clang flag and GCC rejects it outright ("unrecognized command-line option"). +# +# Note `CC ?= clang` does NOT work here. GNU make pre-defines CC = cc, and `?=` +# only assigns when $(origin CC) is `undefined` — a built-in default has origin +# `default`, so `?=` is a no-op and the build silently uses cc. That is fine +# wherever cc happens to be clang (macOS) and fails everywhere else. +# +# Testing $(origin CC) explicitly overrides only make's built-in, while still +# honouring `make CC=...` (origin `command line`) and an exported CC (origin +# `environment`). +ifeq ($(origin CC),default) +CC := clang +endif + +INC := -I$(JBPF_OUT_DIR)/inc \ + -I$(JBPF_DIR)/src/core \ + -I$(JBPF_DIR)/src/common \ + -I$(OCUDU_JANUS_INC) \ + -I$(OCUDU_SPECS_INC) \ + -I$(CONTRACT_INC) + +# -target bpf: codelets are eBPF objects, loaded by ubpf in the gNB. +# +# -D__x86_64__ / -DJBPF_DEBUG_ENABLED / -DJBPF_EXPERIMENTAL_FEATURES are carried +# over from the SDK flags deliberately: the goal is a rebuilt .o that behaves +# identically to the one currently deployed, so this is not the change to +# combine with a flag cleanup. Dropped from the SDK set: -DFMT_USE_BITINT=0 and +# -Wno-reorder-init-list, which existed for the SDK's C++ codelets pulling in +# fmt — ours are plain C. +CFLAGS ?= -O2 -target bpf -Wall \ + -D__x86_64__ \ + -DJBPF_EXPERIMENTAL_FEATURES \ + -DJBPF_DEBUG_ENABLED + +C_SOURCES := $(wildcard *.c) +C_OBJECTS := $(C_SOURCES:.c=.o) + +# ---- diagnostics ---- +# +# Every include root is derived from E3_ROOT, so one wrong assumption about the +# layout shows up as a bare "'jbpf_defs.h' file not found" with no hint about +# which root was wrong. These targets make the resolution visible and check it +# before invoking the compiler. + +.PHONY: show-config check-env + +show-config: + @echo "CODELETS_DIR = $(CODELETS_DIR)" + @echo "E3_ROOT = $(E3_ROOT)" + @echo "CC = $(CC) [origin: $(origin CC)]" + @echo "JBPF_DIR = $(JBPF_DIR)" + @echo "JBPF_OUT_DIR = $(JBPF_OUT_DIR)" + @echo "OCUDU_DIR = $(OCUDU_DIR)" + @echo "OCUDU_JANUS_INC = $(OCUDU_JANUS_INC)" + @echo "CONTRACT_INC = $(CONTRACT_INC)" + @echo "E3_VERIFIER = $(E3_VERIFIER)" + @echo + @echo "include roots:" + @for d in $(patsubst -I%,%,$(INC)); do \ + if [ -d "$$d" ]; then echo " [ok] $$d"; else echo " [MISSING] $$d"; fi; \ + done + @echo + @echo "key headers:" + @for h in $(JBPF_DIR)/src/common/jbpf_defs.h \ + $(JBPF_DIR)/src/core/jbpf_helper.h \ + $(OCUDU_JANUS_INC)/jbpf_srsran_contexts.h \ + $(CONTRACT_INC)/jbpf_srsran_contexts.h; do \ + if [ -f "$$h" ]; then echo " [ok] $$h"; else echo " [missing] $$h"; fi; \ + done + +# Hard-fail on the two roots that are not optional, naming them. +check-env: + @if [ ! -f "$(JBPF_DIR)/src/common/jbpf_defs.h" ]; then \ + echo "ERROR: jbpf sources not found at JBPF_DIR=$(JBPF_DIR)" >&2; \ + echo " Expected $(JBPF_DIR)/src/common/jbpf_defs.h" >&2; \ + if [ -d "$(JBPF_DIR)" ] && [ -z "$$(ls -A $(JBPF_DIR) 2>/dev/null)" ]; then \ + echo " That directory is EMPTY — the submodule is not checked out:" >&2; \ + echo " git -C $(E3_ROOT) submodule update --init --recursive" >&2; \ + else \ + echo " Override it: make JBPF_DIR=/path/to/jbpf" >&2; \ + fi; \ + echo " Run 'make show-config' to see every resolved path." >&2; \ + exit 1; \ + fi + @if [ ! -f "$(OCUDU_JANUS_INC)/jbpf_srsran_contexts.h" ] && \ + [ ! -f "$(CONTRACT_INC)/jbpf_srsran_contexts.h" ]; then \ + echo "ERROR: hook context contract not found." >&2; \ + echo " Tried $(OCUDU_JANUS_INC)/jbpf_srsran_contexts.h" >&2; \ + echo " and $(CONTRACT_INC)/jbpf_srsran_contexts.h" >&2; \ + echo " Pass OCUDU_DIR=/path/to/ocudu-e3, or 'make import-contract'." >&2; \ + exit 1; \ + fi + +# Subdirectories holding a codelet (used by the top-level codelets/Makefile). +CODELET_DIRS := $(patsubst %/Makefile,%,$(wildcard */Makefile)) diff --git a/codelets/README.md b/codelets/README.md index 83a856a..e45af74 100644 --- a/codelets/README.md +++ b/codelets/README.md @@ -1,5 +1,86 @@ -# How to build your codelet +# Codelets -TBD: A script to build codelets inside this project will be released soon. +The jbpf codelets E3Controller loads into the RAN. This directory is the +**canonical home** for them — they are built and verified here, with no +dependency on `jrtc-apps`. -Right now you can follow the guidelines provided in [this repository](https://github.com/microsoft/jrtc-apps). \ No newline at end of file +| Directory | Hook | Program type | Service model | +|---|---|---|---| +| `ecpri_iq_samples/` | `capture_xran_packet` | `jbpf_ran_ofh` | Spectrum (RAN function 1) | +| `uplink_slot_samples/` | `capture_uplink_slot` | `jbpf_ran_slot` | L1 (RAN function 2) | + +## Build + +Prerequisites: `clang` with the BPF target, plus a built `e3_verifier_cli` +(produced by the top-level `./build.sh`, from `codelets/verifier/`). + +```sh +make # build + verify every codelet +make verify # re-verify existing objects +make clean +``` + +To build against an ocudu checkout that is not a sibling of this repo: + +```sh +make OCUDU_DIR=/path/to/ocudu-e3 +``` + +## Verification is mandatory + +`make` fails if a codelet does not verify. That is deliberate, and it is not +belt-and-braces: **the gNB never verifies codelets.** jbpf's load path is +`ubpf_load_elf_ex` followed by `ubpf_compile` +(`external/jbpf/src/core/jbpf.c:968,976`), PREVAIL is not on it, and ubpf's JIT +emits no memory bounds checks at all — only its interpreter does, and jbpf uses +the JIT. So this build step is the only memory-safety gate these codelets ever +pass through, and an advisory one would be decorative. + +The verifier is `codelets/verifier/e3_verifier_cli.cpp`. It replaces the SDK +image's `srsran_verifier_cli`, which models only `jbpf_ran_ofh_ctx` (27 bytes) +and therefore cannot verify the per-slot codelet at all — that is the origin of +the `Upper bound must be at most 27` message. + +## The context contract + +The hook context structs are defined by the **producer**, so ocudu owns them: +`ocudu-e3/include/ocudu/janus/jbpf_srsran_contexts.h`. `Makefile.defs` +puts that directory on the include path; nothing here keeps a maintained copy. + +```sh +make import-contract # take a local copy (for builds with no ocudu checkout) +make check-contract # fail if the local copy has drifted from ocudu +``` + +`check-contract` runs as part of `make`. If you change a ctx struct on the RAN +side, **rebuild every codelet**: a changed layout shifts every offset they read +through, and `e3_verifier_cli` carries `static_assert`s on the ctx extents that +will fail the build first. + +## Deployment + +The codeletset YAMLs resolve the object path through `$JBPF_CODELETS`: + +```yaml +codelet_path: ${JBPF_CODELETS}/uplink_slot_samples/uplink_slot_collect.o +``` + +That variable must point at **this** directory in the deployment environment, +since it is resolved by the LCM load request at runtime: + +```sh +export JBPF_CODELETS=/path/to/E3Controller/codelets +``` + +## Extension IDs + +Helper and program-type IDs live in `include/jbpf_e3_ids.h`, allocated from +jbpf's documented extension ranges (`CUSTOM_HELPER_START_ID`, +`CUSTOM_PROGRAM_START_ID`). That header is shared by the codelets, the +verifier, and the gNB-side helper registration in ocudu — if those three ever +disagree, a codelet that verifies cleanly still fails to load with +`call to nonexistent function `. + +Never add IDs by editing jbpf's core `enum jbpf_helper_type`: that claims an ID +upstream may later allocate to something else, which would silently bind a +codelet call to the wrong native function. diff --git a/codelets/ecpri_iq_samples/ecpri_iq_collect.c b/codelets/ecpri_iq_samples/ecpri_iq_collect.c index de06aea..ed7b78c 100644 --- a/codelets/ecpri_iq_samples/ecpri_iq_collect.c +++ b/codelets/ecpri_iq_samples/ecpri_iq_collect.c @@ -15,9 +15,9 @@ #include "jbpf_defs.h" #include "jbpf_helper.h" -#include "../utils/misc_utils.h" -#include "../utils/net_utils.h" -#include "../xran_packets/xran_format.h" +#include "misc_utils.h" +#include "net_utils.h" +#include "xran_format.h" #include "jbpf_srsran_contexts.h" #include "ecpri_iq_data.h" @@ -83,6 +83,33 @@ struct jbpf_load_map_def SEC("maps") prb_filter_state = { #define NUM_SYMBOLS 14 #define SYMBOL_FLOOR (SLOT_SYMBOLS - NUM_SYMBOLS) +/* Absolute-slot filter (compile-time, edit-and-rebuild). + * At 30 kHz SCS the eCPRI radio-app header encodes slot-within-subframe + * in slotId (0..1) and subframe (0..9); combine them into slot-in-frame + * (0..19) via subframe*2 + slot. Set to -1 to disable and pass every + * slot. Configured for the 7D-S-2U TDD pattern: full-UL slots are + * {8,9,18,19}; slot 19 is the tail UL of the second half-frame. Keeping + * one slot per frame cuts arrival rate to 100 slots/s from ~2000. */ +#define TARGET_ABS_SLOT 19 +#define SLOTS_PER_SUBFRAME 2 /* 30 kHz SCS, numerology mu=1 */ + +/* Exact-symbol filter (compile-time, edit-and-rebuild). Stacks on top + * of TARGET_ABS_SLOT: after slot 19 is picked, keep only symbol + * TARGET_SYMBOL_ID (0..13). Set to -1 to fall back to the range-based + * SYMBOL_FLOOR filter (keep the last NUM_SYMBOLS symbols). Symbol 13 is + * the tail UL symbol, chosen so the RU has finished delivering the full + * slot's payload by the time we sample. Combined with slot 19 this + * yields 1 codelet fire per antenna per frame = 400/s at 4T. */ +#define TARGET_SYMBOL_ID 13 + +/* Single-antenna filter (compile-time, edit-and-rebuild). eCPRI's + * xtc_id field carries the "physical channel ID" (a.k.a. eAxC ID); in + * the Foxconn 4T4R config the UL antennas are ul_port_id [0,1,2,3]. + * Set to -1 to keep all antennas. Stacks on top of slot+symbol filters, + * so with 7D-S-2U + slot 19 + symbol 13 + one antenna the arrival rate + * collapses to 1 codelet fire per frame (100/s). */ +#define TARGET_EAXC -1 + /* ---- Main codelet entry ---- */ @@ -137,6 +164,20 @@ uint64_t jbpf_main(void *state) return JBPF_CODELET_SUCCESS; } + /* --- Single-antenna filter (early drop) --- + * ecpri_xtc_id is __u16 in network byte order; carries the eAxC ID + * that maps to antenna port (ul_port_id in gnb config). Filter here + * before the app-hdr / section-hdr parses, so wrong-antenna packets + * exit with a byte-swap + compare. */ +#if TARGET_EAXC >= 0 + { + uint16_t eaxc = jbpf_ntohs(ecpri_hdr->ecpri_xtc_id); + if (eaxc != TARGET_EAXC) { + return JBPF_CODELET_SUCCESS; + } + } +#endif + /* --- Parse Radio Application Common Header --- */ struct radio_app_common_hdr *app_hdr = (struct radio_app_common_hdr *)next_hdr; if ((void *)(app_hdr + 1) >= pkt_end) { @@ -144,14 +185,35 @@ uint64_t jbpf_main(void *state) } next_hdr = (__u8 *)next_hdr + sizeof(struct radio_app_common_hdr); - /* --- Symbol-count filter (early drop) --- - * sf_slot_sym is 16 bits network-order: [subframeId:4][slotId:6][symbolId:6] - * Keep symbols whose id is in the last NUM_SYMBOLS of the slot. */ + /* --- Symbol/slot filters (early drop) --- + * sf_slot_sym is 16 bits network-order: [subframeId:4][slotId:6][symbolId:6]. + * Order matters here: we drop wrong-slot packets FIRST so a full-frame + * of DL/other UL-slot chatter is discarded before any map lookups or + * verifier-heavier work below. For 7D-S-2U keeping only slot 19 that + * gets us from ~20 slots/frame of UL noise down to 1. */ uint16_t sf_slot_sym = jbpf_ntohs(app_hdr->sf_slot_sym.value); + uint16_t subframe_id_early = (sf_slot_sym >> 12) & 0xF; + uint16_t slot_id_early = (sf_slot_sym >> 6) & 0x3F; + uint16_t abs_slot = subframe_id_early * SLOTS_PER_SUBFRAME + slot_id_early; +#if TARGET_ABS_SLOT >= 0 + if (abs_slot != TARGET_ABS_SLOT) { + return JBPF_CODELET_SUCCESS; + } +#endif uint16_t symbol_id = sf_slot_sym & 0x3F; +#if TARGET_SYMBOL_ID >= 0 + /* Exact-symbol match. Stacks on the slot filter above: any packet + * not carrying symbol TARGET_SYMBOL_ID of slot TARGET_ABS_SLOT gets + * dropped before we touch any maps. */ + if (symbol_id != TARGET_SYMBOL_ID) { + return JBPF_CODELET_SUCCESS; + } +#else + /* Fallback: range-based filter (keep the last NUM_SYMBOLS of the slot). */ if (symbol_id < SYMBOL_FLOOR) { return JBPF_CODELET_SUCCESS; } +#endif /* --- Parse Data Section Header --- */ struct data_section_hdr *data_hdr = (struct data_section_hdr *)next_hdr; @@ -229,6 +291,14 @@ uint64_t jbpf_main(void *state) return JBPF_CODELET_FAILURE; } + /* RAN-side hand-off timestamp piggybacked on ctx->meta_data by the + * hook_capture_xran_packet call site (ofh_message_receiver_impl). + * Read into a dedicated output field so the controller-side ABI is + * self-documenting. Anchor for gnb_to_codelet_us; paired with + * codelet_ts_ns below (stamped just before jbpf_ringbuf_output). + * Held in meta_data because the SDK verifier's OFH ctx descriptor is + * a fixed 27-byte layout — see jbpf_ran_ofh_ctx comment. */ + out->gnb_ts_ns = ctx->meta_data; out->timestamp = entry_ts_us; out->direction = ctx->direction; out->frame_id = app_hdr->frame_id; /* single byte, no endianness issue */ @@ -273,6 +343,12 @@ uint64_t jbpf_main(void *state) out->iq_payload[i] = src[i]; } + /* Codelet-side timestamp captured just before jbpf_ringbuf_output so + * (codelet_ts_ns - gnb_ts_ns) attributes hook -> codelet-dispatch cost, + * matching the uplink_slot_samples pipeline's stage schema. + * Same clock domain as ctx->gnb_ts_ns (CLOCK_REALTIME). */ + out->codelet_ts_ns = jbpf_time_get_ns(); + /* --- Output --- */ int ret = jbpf_ringbuf_output(&output_map, (void *)out, sizeof(struct iq_sample_data)); jbpf_map_clear(&output_tmp_map); diff --git a/codelets/ecpri_iq_samples/ecpri_iq_collect.o b/codelets/ecpri_iq_samples/ecpri_iq_collect.o index 4c956cc..7351fc3 100644 Binary files a/codelets/ecpri_iq_samples/ecpri_iq_collect.o and b/codelets/ecpri_iq_samples/ecpri_iq_collect.o differ diff --git a/codelets/ecpri_iq_samples/ecpri_iq_data.h b/codelets/ecpri_iq_samples/ecpri_iq_data.h index 007ec1e..ef6276e 100644 --- a/codelets/ecpri_iq_samples/ecpri_iq_data.h +++ b/codelets/ecpri_iq_samples/ecpri_iq_data.h @@ -10,7 +10,17 @@ #define MAX_IQ_PAYLOAD_BYTES 8192 struct iq_sample_data { - uint64_t timestamp; /* Timestamp in nanoseconds */ + /* gNB-side hand-off timestamp (CLOCK_REALTIME ns) stamped by the + * ocudu hook caller in ofh_message_receiver_impl::process_new_frame, + * right before hook_capture_xran_packet fires. The RAN anchor for + * RAN -> codelet latency on this pipeline. Same clock domain as + * codelet_ts_ns below and as jbpf_time_get_ns(). */ + uint64_t gnb_ts_ns; + /* Codelet-side timestamp (jbpf_time_get_ns(), CLOCK_REALTIME ns) + * stamped just before jbpf_ringbuf_output. Subtract gnb_ts_ns to + * get gnb_to_codelet_us (hook -> codelet dispatch cost). */ + uint64_t codelet_ts_ns; + uint64_t timestamp; /* Codelet entry timestamp, us mod 2^31 (kept for ASN.1 IQDataIndication.timestamp) */ uint8_t direction; /* 0=DL, 1=UL */ uint8_t frame_id; /* 3GPP frame ID (0-255, wraps every 2.56s) */ uint8_t comp_method; /* Compression method: 0=none, 1=BFP, 2=block scaling, 3=mu-law, 4=modulation */ diff --git a/codelets/include/e3_slot_convert.h b/codelets/include/e3_slot_convert.h new file mode 100644 index 0000000..9306a2f --- /dev/null +++ b/codelets/include/e3_slot_convert.h @@ -0,0 +1,226 @@ +/* + * e3_slot_convert.h — cbf16 -> fp16 row conversion, shared by BOTH writers. + * + * Extracted verbatim from E3Controller's ShmIqWriter::publish_row_cbf16 so the + * controller and the gNB-side jbpf helper produce BYTE-IDENTICAL rows. This is + * the single most important invariant in the publish-direct design: bf16 and + * fp16 are both 2 bytes, so if the two paths ever disagree on the transform or + * the scale, nothing errors — the dApp silently computes a wrong spectrum. + * + * Header-only on purpose: the gNB helper and the controller live in different + * repos and different build systems, so a shared .cpp would need a shared build + * target. See codelets/include/jbpf_e3_slot_api.h for the rest of the contract. + * + * --------------------------------------------------------------------------- + * bf16 vs fp16, and why the scale exists + * + * bf16: 1 sign / 8 exponent / 7 mantissa, bias 127 — the top 16 bits of an + * fp32. Huge range (~3.4e38), coarse precision. + * fp16: 1 sign / 5 exponent / 10 mantissa, bias 15 — max finite 65504, + * min normal ~6.1e-5. Narrow range, finer precision. + * + * bf16 therefore holds values fp16 cannot: above 65504 you get inf, below + * ~6.1e-5 you decay into subnormals or zero. Grid IQ magnitudes depend on gain + * settings, so `scale` shifts them into fp16's representable window before the + * convert. A zero or negative scale zeroes the row and silently breaks the dApp, + * which is why the caller validates it. + * + * The transform is bf16 -> fp32 (a 16-bit shift, free) -> x scale -> + * fp32 -> fp16 (needs rounding + range clamp, hence F16C). + * --------------------------------------------------------------------------- + */ + +#ifndef E3_SLOT_CONVERT_H +#define E3_SLOT_CONVERT_H + +#include +#include + +#if defined(__x86_64__) || defined(_M_X64) +#include +#endif + +/* + * VECTOR PATH REQUIRED. + * + * This loop runs in the gNB's PHY RX thread, inline in the hook. A scalar + * fallback there is not a graceful degradation, it is a latency regression in + * the RAN's slot budget. + * + * This exact trap has already cost us once: the adaptive_cpu dApp's bimodal + * latency turned out to be an x86-scalar fp16 convert, because its fast loop was + * `#if __aarch64__` with no x86 branch and the build carried no -mavx2/-mf16c. + * Nothing warned; it just ran slow, 36% of slots landing in a 600-899us tail. + * + * So: fail the build rather than quietly emit the scalar path. Define + * E3_ALLOW_SCALAR_CONVERT to opt out (host tooling, non-x86, or the + * microbenchmark's deliberate scalar baseline). + */ +#if !defined(E3_ALLOW_SCALAR_CONVERT) +#if !(defined(__AVX2__) && defined(__F16C__)) && !defined(__aarch64__) +#error "e3_slot_convert.h: no vector path. Compile with -mavx2 -mf16c (or -march=native). \ +Define E3_ALLOW_SCALAR_CONVERT only if a scalar convert in the RX thread is genuinely acceptable." +#endif +#endif + +namespace e3_convert { + +/* ---- Row geometry ---- + * + * Must match SharedMemoryHeader's advertised dimensions. Kept as explicit + * parameters rather than compile-time constants so the same code serves 2/4/8 + * ports and different bandwidths (see the plan's D2: strides derived from the + * header at attach time, not from constexpr). + */ +struct RowGeometry { + std::uint32_t nof_symbols; /* 14 */ + std::uint32_t nof_subcarriers; /* 3276 = 273 PRB * 12 */ + std::uint32_t nof_ants; /* row capacity, i.e. antennas the row holds */ + + /* uint16 samples per antenna = 2 * symbols * subcarriers (real + imag). + * Also the per-antenna fp16 stride within the row, so source port p maps to + * destination antenna p. */ + constexpr std::size_t u16_per_ant() const + { + return static_cast(nof_symbols) * nof_subcarriers * 2u; + } + + constexpr std::size_t row_u16() const { return u16_per_ant() * nof_ants; } + constexpr std::size_t row_bytes() const { return row_u16() * sizeof(std::uint16_t); } +}; + +/* ---- Scalar reference ---- + * + * Correctness reference and the microbenchmark's baseline. Not for the RX path. + */ +inline std::uint16_t +bf16_to_fp16_scalar(std::uint16_t bf16, float scale) +{ +#if defined(__F16C__) + union { + std::uint32_t u; + float f; + } u32; + u32.u = static_cast(bf16) << 16; + __m128 ss = _mm_set_ss(u32.f * scale); + __m128i ph = _mm_cvtps_ph(ss, _MM_FROUND_TO_NEAREST_INT); + return static_cast(_mm_extract_epi16(ph, 0)); +#else + union { + std::uint32_t u; + float f; + } u32; + u32.u = static_cast(bf16) << 16; + float v = u32.f * scale; + union { + std::uint32_t u; + float f; + } vu; + vu.f = v; + std::uint32_t sign = (vu.u >> 31) & 0x1u; + std::int32_t exp8 = static_cast((vu.u >> 23) & 0xffu); + std::uint32_t man23 = vu.u & 0x7fffffu; + if (exp8 == 0) { + return static_cast(sign << 15); + } + if (exp8 == 0xff) { + return static_cast((sign << 15) | (0x1fu << 10) | (man23 ? 0x200u : 0u)); + } + std::int32_t exp5 = exp8 - 127 + 15; + if (exp5 >= 0x1f) { + return static_cast((sign << 15) | (0x1fu << 10)); + } + if (exp5 <= 0) { + return static_cast(sign << 15); + } + std::uint32_t man10 = man23 >> 13; + /* round-to-nearest-even on the dropped bits */ + std::uint32_t rem = man23 & 0x1fffu; + if (rem > 0x1000u || (rem == 0x1000u && (man10 & 1u))) { + ++man10; + if (man10 == 0x400u) { + man10 = 0; + ++exp5; + if (exp5 >= 0x1f) { + return static_cast((sign << 15) | (0x1fu << 10)); + } + } + } + return static_cast((sign << 15) | (static_cast(exp5) << 10) | man10); +#endif +} + +/* + * Convert one antenna's worth of bf16 to fp16, applying `scale`. + * + * `n_u16` MUST be a multiple of 8 on the AVX2 path. For 273 PRB / 14 symbols it + * is 91,728 = 8 * 11,466, so there is no tail; asserted by the caller. + */ +inline void +convert_ant(std::uint16_t* dst, const std::uint16_t* src, std::size_t n_u16, float scale) +{ +#if defined(__AVX2__) && defined(__F16C__) + /* 8 bf16 -> 8 fp16 per iteration: + * load 8 bf16 (16 B) -> zero-extend to 8x u32 -> shift left 16 (bf16 + * promoted to fp32) -> multiply by scale -> cvtps_ph -> store 16 B. */ + const __m256 v_scale = _mm256_set1_ps(scale); + std::size_t i = 0; + for (; i + 8 <= n_u16; i += 8) { + __m128i b16 = _mm_loadu_si128(reinterpret_cast(src + i)); + __m256i u32_ext = _mm256_cvtepu16_epi32(b16); + __m256i u32_sh = _mm256_slli_epi32(u32_ext, 16); + __m256 f32 = _mm256_castsi256_ps(u32_sh); + __m256 fscaled = _mm256_mul_ps(f32, v_scale); + __m128i ph = _mm256_cvtps_ph(fscaled, _MM_FROUND_TO_NEAREST_INT); + _mm_storeu_si128(reinterpret_cast<__m128i*>(dst + i), ph); + } + for (; i < n_u16; ++i) { /* tail; empty for 273 PRB */ + dst[i] = bf16_to_fp16_scalar(src[i], scale); + } +#else + for (std::size_t i = 0; i < n_u16; ++i) { + dst[i] = bf16_to_fp16_scalar(src[i], scale); + } +#endif +} + +/* + * Convert a whole slot into one row. + * + * `src` is the gNB resource grid's contiguous cbf16 storage, laid out + * [port][symbol][subcarrier]. `nof_ports` antennas are written; antennas beyond + * that are left untouched (the row is zero-filled once at region creation). + * + * Returns bytes written to `dst`. + */ +inline std::size_t +convert_slot(std::uint16_t* dst, const std::uint16_t* src, const RowGeometry& geom, std::uint32_t nof_ports, + float scale) +{ + std::uint32_t ports = nof_ports ? nof_ports : 1u; + if (ports > geom.nof_ants) { + ports = geom.nof_ants; + } + const std::size_t n = geom.u16_per_ant(); + for (std::uint32_t a = 0; a < ports; ++a) { + convert_ant(dst + static_cast(a) * n, src + static_cast(a) * n, n, scale); + } + return static_cast(ports) * n * sizeof(std::uint16_t); +} + +/* True when a vector path was actually compiled in. Log this at startup: it is + * the difference between ~60us and a latency regression, and it is otherwise + * invisible. */ +inline constexpr bool +has_vector_path() +{ +#if defined(__AVX2__) && defined(__F16C__) + return true; +#else + return false; +#endif +} + +} // namespace e3_convert + +#endif /* E3_SLOT_CONVERT_H */ diff --git a/codelets/include/jbpf_e3_ids.h b/codelets/include/jbpf_e3_ids.h new file mode 100644 index 0000000..67be55d --- /dev/null +++ b/codelets/include/jbpf_e3_ids.h @@ -0,0 +1,84 @@ +/* + * jbpf_e3_ids.h — E3Controller's jbpf extension IDs. + * + * SINGLE SOURCE OF TRUTH for the helper and program-type IDs that + * E3Controller's codelets use. This header is included by: + * + * 1. the codelets (to declare the helper stubs), + * 2. codelets/verifier/e3_verifier_cli.cpp (to register prototypes), + * 3. the gNB-side helper implementation in ocudu-e3 + * (to register the implementations via jbpf_register_helper_function). + * + * If (1)/(2) and (3) ever disagree, a codelet that verifies cleanly will + * still fail to load with "call to nonexistent function ". Keeping all + * three on this header is what prevents that class of skew. + * + * IDs are allocated from jbpf's documented extension ranges + * (external/jbpf/src/common/jbpf_helper_api_defs_ext.h): + * + * CUSTOM_HELPER_START_ID = 32 (MAX_HELPER_FUNC = 64) + * CUSTOM_PROGRAM_START_ID = 8 + * + * We never edit jbpf's core `enum jbpf_helper_type`. Doing so takes an ID + * that upstream may later allocate to something else, which would silently + * bind a codelet call to the wrong native function. + */ + +#ifndef JBPF_E3_IDS_H +#define JBPF_E3_IDS_H + +#include "jbpf_helper_api_defs_ext.h" + +/* ---- Helper IDs ---- */ + +/* + * jbpf_e3_attach(const struct e3_shm_cfg *cfg, uint64_t cfg_len) + * + * Cold path. Attaches the gNB-side helper to the E3Controller-owned + * /e3_ran_buffers region and caches its geometry from SharedMemoryHeader. + */ +#define JBPF_E3_ATTACH_ID (CUSTOM_HELPER_START_ID + 0) + +/* + * jbpf_e3_publish_slot(const void *src, uint64_t src_len, + * const struct e3_slot_sel *sel, uint64_t sel_len) + * + * Hot path. Converts the cbf16 slot to the dApp's fp16 row format and + * writes it into the attached region. Returns + * ((int64_t)fh_buffer_index << 32) | fh_write_index + * or a negative value on error. + * + * The codelet never holds a pointer into the destination region: it is + * derived from geometry the helper alone owns. Only `src` needs to be a + * verifier-proven (ptr, size) pair. + */ +#define JBPF_E3_PUBLISH_SLOT_ID (CUSTOM_HELPER_START_ID + 1) + +/* ---- Program type IDs ---- */ + +/* + * ocudu's jbpf_srsran_contexts.h already allocates + * JBPF_PROG_TYPE_RAN_OFH = CUSTOM_PROGRAM_START_ID + 0 + * JBPF_PROG_TYPE_RAN_LAYER2 = CUSTOM_PROGRAM_START_ID + 1 + * JBPF_PROG_TYPE_RAN_MAC_SCHED = CUSTOM_PROGRAM_START_ID + 2 + * JBPF_PROG_TYPE_RAN_GENERIC = CUSTOM_PROGRAM_START_ID + 3 + * + * The per-slot upper-PHY hook needs its own program type, because its ctx + * (struct jbpf_ran_slot_ctx, 47 B) is larger than jbpf_ran_ofh_ctx (27 B). + * Declaring the slot codelet as SEC("jbpf_ran_ofh") is exactly why + * verification currently fails with "Upper bound must be at most 27". + * + * TODO: this belongs in ocudu's jbpf_srsran_contexts.h next to the other + * four, since program types are part of the RAN-side contract. Defined here + * until that upstream change lands; the value must not collide. + */ +#define JBPF_E3_PROG_TYPE_RAN_SLOT (CUSTOM_PROGRAM_START_ID + 4) + +/* ELF section name the slot codelet is compiled into, and the prefix the + * verifier matches to pick the program type above. jbpf loads codelets by + * *function* name (JBPF_MAIN_FUNCTION_NAME via ubpf_load_elf_ex), not by + * section, so this name is only meaningful to the verifier. */ +#define JBPF_E3_SEC_RAN_SLOT "jbpf_ran_slot" +#define JBPF_E3_SEC_RAN_OFH "jbpf_ran_ofh" + +#endif /* JBPF_E3_IDS_H */ diff --git a/codelets/include/jbpf_e3_slot_api.h b/codelets/include/jbpf_e3_slot_api.h new file mode 100644 index 0000000..e9d08e6 --- /dev/null +++ b/codelets/include/jbpf_e3_slot_api.h @@ -0,0 +1,191 @@ +/* + * jbpf_e3_slot_api.h — wire contract for the per-slot L1 path (RAN function 2). + * + * Shared by FOUR consumers, which is why it lives here rather than in a + * codelet directory: + * + * 1. codelets/uplink_slot_samples/uplink_slot_collect.c (producer) + * 2. the gNB-side helper in ocudu-e3 (executes the copy) + * 3. E3Controller's SlotIqPipeline (reads descriptors) + * 4. codelets/verifier/e3_verifier_cli.cpp (sizes the sel arg) + * + * The IQ payload is NOT carried here. The helper writes it straight into the + * E3Controller-owned /e3_ran_buffers region in the dApp's fp16 format, and the + * codelet publishes only the ~64 B descriptor below. See + * report/plan_codelet_verification_and_config_scaling.md. + */ + +#ifndef JBPF_E3_SLOT_API_H +#define JBPF_E3_SLOT_API_H + +#include + +/* ---- Per-config slot sizes ---- + * + * ports * 14 symbols * 3276 subcarriers * 4 B/sample (cbf16 = bf16 I + bf16 Q). + * 3276 = 273 PRB * 12, i.e. 100 MHz at 30 kHz SCS. + * + * These exist as separate compile-time constants, not a computed value, + * because the verifier requires the length passed to the publish helper to be + * a LITERAL at the call site — see the ladder in uplink_slot_collect.c. + */ +#define E3_SLOT_BYTES_2P 366912u /* 2 ports */ +#define E3_SLOT_BYTES_4P 733824u /* 4 ports — today's deployment */ +#define E3_SLOT_BYTES_8P 1467648u /* 8 ports */ + +/* ---- Selector: what to publish (control input, RAN function 2) ---- + * + * TEMPORAL ONLY in this phase. Spatial selection (port/PRB/symbol subsets) is + * deliberately deferred; `reserved` is where it lands, so adding it does not + * break the ABI. See the plan's Out-of-scope section for why temporal is safe + * for multiple dApps and spatial is not. + * + * Delivered by the E3Controller over a jbpf control-input channel. That is a + * CHANNEL, not a one-shot: the codelet caches the latest value in an array map + * and re-reads the cache on every invocation, so the controller can retune the + * selection at any time as subscriptions come and go. + */ +struct e3_slot_sel { + uint16_t version; /* E3_SLOT_SEL_VERSION */ + + /* Slots per radio frame for the live numerology, used as a sanity check + * against slot_mask's width. 20 at 30 kHz SCS. */ + uint16_t nof_slots_per_frame; + + /* Bit i set => publish slots whose slot_index == i. + * + * ZERO MEANS PUBLISH EVERYTHING. That is the default when no dApp has + * asked for a filter, and it makes behaviour identical to the unfiltered + * system. Filtering is strictly opt-in and can only ever reduce what the + * RAN publishes. + * + * slot_index is the index within the RADIO FRAME + * (ocudu slot_point::slot_index() == count % nof_slots_per_frame), so at + * 30 kHz SCS the valid range is 0..19 — NOT 0..1. Under a 7DS2U TDD + * pattern only {8, 9, 18, 19} are UL data slots, so a mask selecting + * anything else yields no indications at all; the controller validates + * this at subscription time. + * + * 64 bits covers numerology <= 2 (40 slots/frame). Widen for mu > 2. */ + uint64_t slot_mask; + + /* Frame-level decimation: publish only when (sfn % sfn_mod) == sfn_offset. + * sfn_mod 0 or 1 means every frame. Cheapest lever for a dApp that needs + * only occasional slots, and the cheapest way to bound RX-thread cost. */ + uint16_t sfn_mod; + uint16_t sfn_offset; + + /* Reserved for spatial selection (port_mask, symbol range, PRB range). */ + uint8_t reserved[20]; +}; + +#define E3_SLOT_SEL_VERSION 1u + +/* ---- Attach config: where to publish ---- + * + * The E3Controller OWNS /e3_ran_buffers — it creates, sizes and tears down the + * region, and writes the actual grid dimensions into its SharedMemoryHeader. + * The helper only attaches. Delivered via control input so ownership stays with + * the controller and the gNB needs no duplicate configuration. + */ +struct e3_shm_cfg { + uint16_t version; /* E3_SHM_CFG_VERSION */ + uint16_t epoch; /* bumped when the controller recreates the region; lets + * the helper detect and refuse a stale mapping */ + char name[64]; /* POSIX shm name, e.g. "/e3_ran_buffers" */ + + /* Row geometry. The controller owns the region and therefore the geometry; + * the helper cross-checks these against SharedMemoryHeader::num_fh_samples + * on attach and refuses a mismatch rather than writing at the wrong stride. + * + * Carried explicitly because the NVIDIA header advertises only + * num_fh_samples (the whole row's uint16 count), from which symbols and + * subcarriers cannot be recovered independently. */ + uint16_t nof_symbols; /* 14 */ + uint16_t nof_subcarriers; /* 3276 */ + + /* bf16 -> fp16 scale, as applied by ShmIqWriter (E3_CBF16_SCALE). + * + * Pushed down rather than read from the environment on the gNB side. The + * controller owns this calibration, and if the two writers ever disagreed + * on it the rows would differ with no error — bf16 and fp16 are both 2 + * bytes, so a wrong scale is silently wrong data, not a failure. Making it + * part of the contract is what keeps the byte-identical invariant true. */ + float scale; + + uint8_t reserved[16]; +}; + +#define E3_SHM_CFG_VERSION 1u + +/* ---- Descriptor: what was published ---- + * + * One per published slot, through the codelet's jbpf output map. Replaces the + * old ~733 KB `struct uplink_slot_sample`, so the jbpf ring drops from tens of + * MiB to a couple of KiB and stops being a per-config sizing problem. + */ +struct e3_slot_desc { + /* A1 entry: gNB hand-off timestamp, stamped by the hook caller immediately + * before the hook fires - on the last symbol, grid complete, before + * anything has been copied. RAN anchor for end-to-end latency. + * + * CLOCK_MONOTONIC ns, matching jbpf_time_get_ns() (patched) and the + * controller's dispatcher poll, so all three subtract cleanly. */ + uint64_t gnb_ts_ns; + + /* jbpf_time_get_ns() as the FIRST statement of jbpf_main, before any + * control-input handling or publishing. + * + * codelet_entry_ts_ns - gnb_ts_ns = ocudu hook -> codelet entry, i.e. + * ubpf JIT dispatch and jbpf plumbing ONLY. */ + uint64_t codelet_entry_ts_ns; + + /* jbpf_time_get_ns() just before submit; same clock, subtracts cleanly. + * + * codelet_ts_ns - codelet_entry_ts_ns = the codelet's own execution. In + * `writer: gnb` that is dominated by jbpf_e3_publish_slot(): the + * cbf16 -> fp16 convert of the whole grid plus the row write into + * /e3_ran_buffers, ~700 KiB read and ~700 KiB written per slot. It is + * memory-bandwidth-bound and is the larger half by a wide margin. + * + * These were one field. It was documented as "codelet entry" but stamped at + * the end, so when `writer: gnb` moved the data plane into the codelet the + * convert silently folded into what the CSV reported as hook->codelet + * delivery. Splitting them is what makes gnb_to_codelet_us comparable with a + * `writer: controller` run, and with published figures that report delivery + * and shared-memory transfer as separate contributions. */ + uint64_t codelet_ts_ns; + + /* Bytes the helper wrote into the row. 0 with E3_SLOT_FLAG_TRUNCATED set + * means it refused rather than writing a prefix. */ + uint32_t bytes_written; + + /* Where the row is: (fh_buffer_index, fh_write_index) into + * /e3_ran_buffers, exactly as the Indication already carries them. */ + uint32_t fh_write_index; + uint8_t fh_buffer_index; + + uint8_t flags; + uint16_t sfn; + uint16_t subframe_id; + uint16_t slot_id; /* index within the radio frame; see slot_mask above */ + uint16_t sector_id; + + /* Grid shape as OBSERVED, so the consumer can cross-check + * nof_ports * nof_symbols * nof_subcarriers * 4 == bytes_written. */ + uint16_t nof_ports; + uint16_t nof_symbols; + uint16_t nof_subcarriers; + + uint8_t reserved[6]; +}; + +/* No ladder rung matched the observed grid, so nothing was published. Set + * rather than silently writing a prefix: publishing 4 of 8 antennas with no + * signal is a silently wrong measurement, which is worse than a gap. */ +#define E3_SLOT_FLAG_TRUNCATED 0x01u + +/* The selector was active and this slot passed the filter (diagnostic). */ +#define E3_SLOT_FLAG_FILTERED 0x02u + +#endif /* JBPF_E3_SLOT_API_H */ diff --git a/codelets/include/misc_utils.h b/codelets/include/misc_utils.h new file mode 100644 index 0000000..c3d1dc9 --- /dev/null +++ b/codelets/include/misc_utils.h @@ -0,0 +1,32 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. +// +// Vendored into E3Controller/codelets/include/ from the jbpf SDK's codelet +// utils (jrtc-apps/codelets/) so that the codelet build has no dependency on +// that tree. Unmodified apart from this note; MIT license above retained. + +#ifndef MISC_UTILS_H +#define MISC_UTILS_H + +#define JBPF_CTRL_CODELET_SUCCESS (1) +#define JBPF_CODELET_SUCCESS (0) +#define JBPF_CODELET_FAILURE (-1) + + +// Macto to create an explicirt rbid... +// if srb, use rb_id as is, else add 10 to the rb_id. +// so srb=true rb_id-1, will reults in 1 +// and srb=false rb_id-1, will result in 11. +#define RBID_2_EXPLICIT(is_srb, rb_id) ((is_srb) ? (rb_id) : ((rb_id) + 10)) + +#define RBID_FROM_EXPLICIT(explicit_rb_id, is_srb, rb_id) { \ + if (explicit_rb_id < 10) { \ + is_srb = true; \ + rb_id = explicit_rb_id; \ + } else { \ + is_srb = false; \ + rb_id = explicit_rb_id - 10; \ + } \ +} +#endif // MISC_UTILS_H + diff --git a/codelets/include/net_utils.h b/codelets/include/net_utils.h new file mode 100644 index 0000000..96a7065 --- /dev/null +++ b/codelets/include/net_utils.h @@ -0,0 +1,26 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. +// +// Vendored into E3Controller/codelets/include/ from the jbpf SDK's codelet +// utils (jrtc-apps/codelets/) so that the codelet build has no dependency on +// that tree. Unmodified apart from this note; MIT license above retained. + +#ifndef NET_UTILS_H +#define NET_UTILS_H + +#include + +#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__ +#define jbpf_ntohs(x) __builtin_bswap16(x) +#define jbpf_htons(x) __builtin_bswap16(x) +#define jbpf_ntohl(x) __builtin_bswap32(x) +#define jbpf_htonl(x) __builtin_bswap32(x) +#elif __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +#define jbpf_ntohs(x) (x) +#define jbpf_htons(x) (x) +#define jbpf_ntohl(x) (x) +#define jbpf_htonl(x) (x) +#endif + +#endif // NET_UTILS_H + diff --git a/codelets/include/xran_format.h b/codelets/include/xran_format.h new file mode 100644 index 0000000..b34cf0d --- /dev/null +++ b/codelets/include/xran_format.h @@ -0,0 +1,156 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. +// +// Vendored into E3Controller/codelets/include/ from the jbpf SDK's codelet +// utils (jrtc-apps/codelets/) so that the codelet build has no dependency on +// that tree. Unmodified apart from this note; MIT license above retained. + +#ifndef XRAN_FORMAT_H_ +#define XRAN_FORMAT_H_ + +#include +#include + +struct vlan_hdr { + __be16 h_vlan_TCI; /* VLAN Tag Control Information */ + __be16 h_vlan_encapsulated_proto; /* Ethernet protocol type */ +}; + + +#define ECPRI_IQ_DATA 0x00 // eCPRI Message Type for IQ Data +#define ECPRI_RT_CONTROL_DATA \ + 0x02 // eCPRI Message Type for Real-Time Control Data + +struct prb_stats_event +{ + __u32 num_prbs_used; + __u32 total_num_prbs; +}; + +union xran_ecpri_cmn_hdr +{ + struct + { + __u8 ecpri_concat:1; + __u8 ecpri_resv:3; + __u8 ecpri_ver:4; + __u8 ecpri_mesg_type; + __u16 ecpri_payl_size; + } bits; + struct + { + __u32 data_num_1; + } data; +} __attribute__((packed)); + +union ecpri_seq_id +{ + struct + { + __u8 seq_id:8; + __u8 sub_seq_id:7; + __u8 e_bit:1; + } bits; + struct + { + __u16 data_num_1; + } data; +} __attribute__((packed));; + +struct xran_ecpri_hdr +{ + union xran_ecpri_cmn_hdr cmnhdr; + __u16 ecpri_xtc_id; + union ecpri_seq_id ecpri_seq_id; +} __attribute__((packed)); + +struct radio_app_common_hdr +{ + /* Octet 9 */ + union { + __u8 value; + struct { + __u8 filter_id:4; /**< This parameter defines an index to the channel filter to be + used between IQ data and air interface, both in DL and UL. + For most physical channels filterIndex =0000b is used which + indexes the standard channel filter, e.g. 100MHz channel filter + for 100MHz nominal carrier bandwidth. (see 5.4.4.3 for more) */ + __u8 payl_ver:3; /**< This parameter defines the payload protocol version valid + for the following IEs in the application layer. In this version of + the specification payloadVersion=001b shall be used. */ + __u8 data_direction:1; /**< This parameter indicates the gNB data direction. */ + }; + }data_feature; + + /* Octet 10 */ + __u8 frame_id:8; /**< This parameter is a counter for 10 ms frames (wrapping period 2.56 seconds) */ + + /* Octet 11 */ + /* Octet 12 */ + union { + __u16 value; + struct { + __u16 symb_id:6; /**< This parameter identifies the first symbol number within slot, + to which the information of this message is applies. */ + __u16 slot_id:6; /**< This parameter is the slot number within a 1ms sub-frame. All slots in + one sub-frame are counted by this parameter, slotId running from 0 to Nslot-1. + In this version of the specification the maximum Nslot=16, All + other values of the 6 bits are reserved for future use. */ + __u16 subframe_id:4; /**< This parameter is a counter for 1 ms sub-frames within 10ms frame. */ + }; + }sf_slot_sym; + +} __attribute__((packed)); + +struct compression_hdr +{ + __u8 ud_comp_meth:4; + /**< udCompMeth| compression method |udIqWidth meaning + ---------------+-----------------------------+-------------------------------------------- + 0000b | no compression |bitwidth of each uncompressed I and Q value + 0001b | block floating point |bitwidth of each I and Q mantissa value + 0010b | block scaling |bitwidth of each I and Q scaled value + 0011b | mu-law |bitwidth of each compressed I and Q value + 0100b | modulation compression |bitwidth of each compressed I and Q value + 0100b - 1111b | reserved for future methods |depends on the specific compression method + */ + __u8 ud_iq_width:4; /**< Bit width of each I and each Q + 16 for udIqWidth=0, otherwise equals udIqWidth e.g. udIqWidth = 0000b means I and Q are each 16 bits wide; + e.g. udIQWidth = 0001b means I and Q are each 1 bit wide; + e.g. udIqWidth = 1111b means I and Q are each 15 bits wide + */ +} __attribute__((packed));; + +struct data_section_compression_hdr +{ + struct compression_hdr ud_comp_hdr; + __u8 rsrvd; /**< This parameter provides 1 byte for future definition, + should be set to all zeros by the sender and ignored by the receiver. + This field is only present when udCompHdr is present, and is absent when + the static IQ format and compression method is configured via the M-Plane */ + + /* TODO: support for Block Floating Point compression */ + /* udCompMeth 0000b = no compression absent*/ +}; + +struct data_section_hdr +{ + union { + __u32 all_bits; + struct { + __u32 num_prbu:8; /**< 5.4.5.6 number of contiguous PRBs per control section */ + __u32 start_prbu:10; /**< 5.4.5.4 starting PRB of control section */ + __u32 sym_inc:1; /**< 5.4.5.3 symbol number increment command XRAN_SYMBOLNUMBER_xxxx */ + __u32 rb:1; /**< 5.4.5.2 resource block indicator, XRAN_RBIND_xxx */ + __u32 sect_id:12; /**< 5.4.5.1 section identifier */ + }; + }fields; +} __attribute__((packed)); + +struct compression_params +{ + __u8 exponent:4; + __u8 reserved:4; +}__attribute__((packed)); + +#endif \ No newline at end of file diff --git a/codelets/uplink_slot_samples/README.md b/codelets/uplink_slot_samples/README.md new file mode 100644 index 0000000..bdb8bbd --- /dev/null +++ b/codelets/uplink_slot_samples/README.md @@ -0,0 +1,217 @@ +# `uplink_slot_samples` — what this codelet gathers, and what the numbers mean + +This codelet is the RAN-side entry point of the E3 uplink IQ path. It runs in the +gNB's PHY RX thread, under ubpf, on every completed uplink slot. + +It is **descriptor-only**. No IQ ever crosses the jbpf ring: the codelet calls the +gNB-side helper, which converts and writes the samples straight into +`/e3_ran_buffers`, and the codelet then publishes a ~64-byte descriptor saying +*where the row landed*. That is the whole point of the design — it halves total +memory traffic versus copying the grid through jbpf, and it is why +`iq_bytes` reads `0` in the controller's stage CSV. + +--- + +## 1. What the codelet reads (input) + +`struct jbpf_ran_slot_ctx` — defined by the hook PRODUCER, i.e. ocudu, in +`include/ocudu/janus/jbpf_srsran_contexts.h`. That file is the single source of +truth; `codelets/Makefile` diffs it at build time (`check-contract`) and fails on +drift, because a field added on the RAN side would silently shift every offset the +codelet reads through. + +| field | meaning | +|---|---| +| `data` / `data_end` | packet-style bounds over the cbf16 resource grid | +| `gnb_ts_ns` | `CLOCK_MONOTONIC` ns, stamped by the gNB just before the hook fires | +| `sfn`, `subframe_id`, `slot_id` | slot identity (`slot_id` = index within the 10 ms frame) | +| `ctx_id` | sector | +| `nof_ports`, `nof_symbols`, `nof_subcarriers` | grid shape for THIS slot | + +The grid is `nof_ports x nof_symbols x nof_subcarriers` × `cbf16_t` (4 B), laid +out `[port][symbol][subcarrier]`. For the 4×4 / 273 PRB / 30 kHz configuration +that is **733,824 bytes per slot**. + +## 2. What the codelet publishes (output) + +`struct e3_slot_desc` — the shared contract in `codelets/include/jbpf_e3_slot_api.h`, +vendored byte-identically into the gNB at `ocudu_janus/e3/`. Both copies are +compared at configure time; a divergence would make the two writers disagree with +**no runtime error**, because bf16 and fp16 are both 2 bytes. + +| field | meaning | +|---|---| +| `gnb_ts_ns` | copied through from the ctx (the RAN-side anchor) | +| `codelet_ts_ns` | `jbpf_time_get_ns()` at the **END** of the codelet — see §3 | +| `bytes_written` | bytes the helper actually wrote; `0` means it refused | +| `fh_buffer_index`, `fh_write_index` | the row the helper wrote — what the Indication carries | +| `flags` | `E3_SLOT_FLAG_TRUNCATED` when the helper refused rather than writing a partial row | +| `sfn`, `subframe_id`, `slot_id`, `sector_id` | slot identity | +| `nof_ports`, `nof_symbols`, `nof_subcarriers` | grid shape, so the controller can validate against its config | + +## 3. The timestamps — and the one that is easy to misread + +`codelet_ts_ns` is stamped at `uplink_slot_collect.c:220`, **after** the +`jbpf_e3_publish_slot()` call at `:174-180`. So the controller's +`gnb_to_codelet_us` column is **not** "hook → codelet entry", despite what its +name suggests. It spans: + +``` +gnb_ts_ns clock_gettime(CLOCK_MONOTONIC) immediately before the hook fires, + on the LAST symbol of the slot, i.e. when the resource grid is + complete (lib/phy/upper/upper_phy_rx_symbol_handler_impl.cpp:73) + | + | ubpf JIT dispatch + the codelet's constant-size bounds-check ladder + | jbpf_e3_publish_slot(): bf16 -> fp16 AVX2/F16C convert + | + the full 733,824-byte row write into /e3_ran_buffers + | descriptor field writes + v +codelet_ts_ns jbpf_time_get_ns() (uplink_slot_collect.c:220) +``` + +**Read it as "gNB hands off the completed slot → the IQ is in shared memory."** +It excludes PHY symbol processing (the anchor is grid-complete) and it is *not* +jbpf overhead. + +Sanity check on that reading: the convert alone accounts for the large majority of +this stage, with only a few microseconds of jbpf dispatch left over. So treat this +column as a memory-bandwidth measurement, not as instrumentation overhead. + +### One clock, everywhere + +Every stamp on this path is **`CLOCK_MONOTONIC`**, and that is a five-place +invariant rather than a local choice: + +| site | file | +|---|---| +| `gnb_ts_ns` | `lib/phy/upper/upper_phy_rx_symbol_handler_impl.cpp` | +| `codelet_ts_ns`, via `jbpf_time_get_ns()` | `jbpf/src/core/jbpf_helper_impl.c` — **local patch**, see `jbpf_monotonic_time.patch` | +| `dispatch_ts_ns` | `E3Controller/src/e3sm/slot_iq_pipeline.cpp` | +| handler entry | `E3Controller/src/e3sm/l1_kpm/e3sm_layer_1.cpp` | +| the dApp's arrival age | `adaptive_cpu/subcarrier_power_app.cpp` (`now_mono_ns`) | + +Change one without the others and the subtraction mixes two unrelated epochs. +**That failure is silent** — no error, just plausible-looking wrong latencies, or, +in unsigned arithmetic, a wrap to ~1.8e19 that clamps to 0. It has already +happened once, when the dApp measured a `CLOCK_REALTIME` wire timestamp against +`CLOCK_MONOTONIC` and reported an arrival age of ~0. + +Two reasons for monotonic over realtime: + +- it is the clock **latrec** stamps, so the stage CSVs and the latrec rings share + one epoch and join directly, with no offset arithmetic; +- `CLOCK_REALTIME` can be stepped by NTP, and a step mid-run corrupts every stage + derived from these absolute stamps. + +The cost: `CLOCK_MONOTONIC` shares an origin only within a single boot. These +stamps are comparable across **processes on one host**, not across hosts. A +distributed deployment would have to move back to an NTP-disciplined clock — and +move all five sites together. + +## 4. Downstream: the controller's stage CSV + +Enabled by `logging.stats_log_path` in `e3_controller.yaml`. One row per +**published** slot: + +``` +slot_seq,gnb_to_codelet_us,codelet_to_dispatch_us,dispatch_to_handler_us, +shm_ns,encode_ns,emit_ns,nof_subc,iq_bytes +``` + +| column | meaning | +|---|---| +| `gnb_to_codelet_us` | §3 — the data plane: hook → IQ in SHM | +| `codelet_to_dispatch_us` | codelet → controller's jbpf dispatcher poll | +| `dispatch_to_handler_us` | dispatcher → `on_sample` (SPSC queue wait) | +| `shm_ns` | controller-side convert cost. **0 in `writer: gnb` mode** — the helper already wrote the row | +| `encode_ns` | SM payload encode (APER or JSON) | +| `emit_ns` | `emit_outbound` fan-out, i.e. SM-side enqueue only | +| `iq_bytes` | **0** in descriptor-only mode — confirms no IQ crossed jbpf | + +A sibling `_drops.csv` carries cumulative drop counts per reason +(`no_subscribers`, `ran_published_nothing`, `blob_too_small`, …), one row per +second, written only when a total changes. + +## 5. Measured results + +Concrete measurements are deliberately **not** kept in this file. They are +run-, pod- and configuration-specific, and this README is publishable +documentation of *what* is measured rather than a results record. + +Current numbers live in the internal reports: + +- `report/e2e_latency_audit_7ds2u.md` — 7DS2U, 4x4, 273 PRB, 30 kHz +- `report/e2e_latency_audit_2ds7u.md` — the higher-UL-rate baseline + +Those cover the per-stage percentiles, the `gnb_to_codelet_us` bimodality and its +occupancy explanation, the dApp end-to-end figure, and the loss map below filled +in with actual counts. + +## 6. Known blind spots in the accounting + +A low drop count in `_drops.csv` does **not** mean nothing was lost end to end. +Each segment below is counted by a different thing, and one segment is counted by +nothing at all. + +| segment | who counts it | +|---|---| +| RAN → controller (hook fired vs codelet published vs ring delivered) | **nobody** — no counter exists on either side | +| pipeline SPSC queue full | `SlotIqPipeline::dropped_samples()`, printed **only** in the controller's shutdown summary, never in a CSV | +| controller SM (9 named reasons) | `_drops.csv` + the shutdown summary | +| libe3 outbound queue | libe3's own log / its `L9_DROP` latrec stage | +| wire → dApp | the dApp only (`life: rx/drop`) | +| dApp shed (`no_range`, `shed_age`, …) | the dApp's `[Status]` block | + +Practical consequences: capture the controller's **stdout** (the queue-full count +lives only there), and compare the RAN's expected UL-slot rate against the +published rate by hand — a mismatch there is invisible to every counter listed +above. Expected rate is +`nrofUplinkSlots / dl-UL-TransmissionPeriodicity` (both visible in the gNB's +resolved `pattern1` log line), times sectors. + +Two consequences: + +1. The controller's drop counters only see slots that reach `on_sample`. Anything + lost in the hook, the codelet, or the jbpf ring is invisible to them. +2. **Sequence-gap detection is deliberately NOT implemented.** Under TDD only UL + slots fire the hook, and with `allow_request_on_empty_uplink_slot` an idle UE + can leave UL slots unprocessed too — so a missing slot index is not + necessarily a loss, and inferring drops from gaps would be confidently wrong. + Doing it properly needs a learned UL-slot mask. + +## 7. Building and verifying + +```bash +make -B OCUDU_DIR=/path/to/ocudu-e3 # -B is REQUIRED, see below +``` + +Every build compiles **and then verifies** with the offline PREVAIL verifier +(`codelets/verifier/`), and a verification failure fails the build, leaving the +previous good `.o` in place. Current size: **346 instructions**, verified. + +`make clean` deletes only `*.o.tmp` **by design** — the `.o` files are tracked and +deployed, so a failed compile must not destroy the last good one. The consequence +is that a bare `make` against an unchanged `.c` is a silent **no-op**; use `-B` to +force a real rebuild. + +Verification matters here because the gNB does **no** load-time verification: it +loads via ubpf, whose JIT emits no memory bounds checks. This build step is the +only safety gate that ever runs on this object. That is also why the publish +helper re-validates the source window at runtime (`e3_set_src_window`) instead of +trusting that the loaded object was verified. + +## 8. Enabling end-to-end latency tracing + +The stage CSV stops at `emit`, so the `emit → wire → dApp receive` segment is +invisible to it — and in practice that segment dominates. To see inside it, use +latrec — +libe3 and the dApp share the format, so their rings correlate: + +```bash +. run_scripts/latrec_capture.sh /tmp/latrec/run1 # in EVERY pane, same dir +# start controller, then dApp +python3 /workspace/e3_improved/libe3/tools/latrec2csv.py /tmp/latrec/run1 -o /tmp/latrec/run1/out +``` + +Both are no-ops unless `LATREC_DIR` is set, so production pays nothing. Mind the +ring defaults: several roles default to 2^22 records = 128 MiB each. diff --git a/codelets/uplink_slot_samples/uplink_slot_collect.c b/codelets/uplink_slot_samples/uplink_slot_collect.c index df10226..918dcf1 100644 --- a/codelets/uplink_slot_samples/uplink_slot_collect.c +++ b/codelets/uplink_slot_samples/uplink_slot_collect.c @@ -2,26 +2,39 @@ * uplink_slot_collect.c * * Codelet attached to the ocudu hook `capture_uplink_slot` (defined in - * upper_phy_rx_symbol_handler_impl.cpp). Fires once per UL slot, on the - * last symbol, with the resource grid's contiguous cbf16_t storage - * already pointed at via the hook ctx. + * upper_phy_rx_symbol_handler_impl.cpp). Fires once per UL slot, on the last + * symbol, with the resource grid's contiguous cbf16_t storage pointed at via + * the hook ctx. * - * The ocudu hook already hands us the - * fully-assembled slot in resource-grid order. We do one memcpy from - * the gnb-owned grid storage into the output ring slot (jbpf SHM the - * E3Controller polls), stamp metadata, and submit. + * WHAT THIS CODELET DOES *NOT* DO ANY MORE: copy the grid. * - * Output layout: see uplink_slot_data.h. + * It previously moved ~733 KB into a jbpf output-map slot. eBPF cannot do that + * job well — the in-VM copy ran at ~8.5 GB/s, and the ISA has no + * floating-point or SIMD instructions at all (114 integer opcodes), so it also + * cannot perform the cbf16 -> fp16 conversion the dApp's row format needs. + * + * So the split is now: + * - this codelet decides *whether* to publish (verified bytecode, policy), + * - a native SIMD helper does the convert + row write (mechanism). + * + * The codelet never holds a pointer into the destination: it passes a selector + * and receives back the (buffer, row) indices the helper used. Only the SOURCE + * needs verifier-proven bounds, which shrinks the trusted surface considerably. + * + * Output: one ~64 B `struct e3_slot_desc` per published slot (see + * codelets/include/jbpf_e3_slot_api.h), which the E3Controller turns into an + * E3 Indication. The IQ itself goes straight to /e3_ran_buffers. */ #include "jbpf_defs.h" #include "jbpf_helper.h" #include "jbpf_srsran_contexts.h" -#include "uplink_slot_data.h" -/* Codelet return codes - mirror the macros in jrtc-apps/codelets/utils/ - * misc_utils.h so we don't take an external dependency on the utils - * tree just for these two constants. */ +#include "jbpf_e3_ids.h" +#include "jbpf_e3_slot_api.h" + +/* Codelet return codes. Defined locally so this codelet takes no dependency on + * an external SDK utils tree for two constants. */ #ifndef JBPF_CODELET_SUCCESS #define JBPF_CODELET_SUCCESS (0) #endif @@ -29,60 +42,176 @@ #define JBPF_CODELET_FAILURE (-1) #endif -/* ---- Maps ---- - * One output map, zero-copy via jbpf_get_output_buf / jbpf_send_output. - * Using jbpf_output_map rather than jbpf_ringbuf_map so the codelet - * writes the full ~733 KB slot struct (4 antennas x 14 sym x 3276 subc x - * cbf16_t) directly into the ring slot without going through a temp-map - * intermediate first. One memcpy on the codelet side instead of two. - */ -jbpf_output_map(output_map, struct uplink_slot_sample, 32); - -/* ---- Constants ---- */ - -/* Chunked copy size. 128 bytes = two cache lines = 32 cbf16_t samples. - * Picked so that: - * - Constant size lets __builtin_memcpy lower to a short fixed - * vector-move sequence the eBPF verifier handles cheaply. - * - 128 divides the full 4-antenna slot payload - * (4 * 14 * 3276 * sizeof(cbf16_t) = 733 824 bytes; 733824 / 128 = - * 5 733 exactly), so the flat slot copy leaves no trailing tail for - * the 273-PRB/4-port deployment. (A grid whose payload is not a - * multiple of 128 would leave up to 127 bytes uncopied at the tail.) - * - Outer loop bound = MAX_SLOT_IQ_BYTES / 128 = 5 733 iterations, - * half the per-chunk bounds checks of the previous 64 B chunks - * (~11 466). The copy is memory-bandwidth bound, so this only trims - * the loop overhead, not the bulk data-movement cost. +/* ---- Helper stubs ---- + * + * IDs come from jbpf_e3_ids.h, which the gNB-side registration also includes — + * that shared header is what stops a codelet from verifying cleanly and then + * failing to load with "call to nonexistent function". + * + * IDs are in jbpf's documented CUSTOM_HELPER_START_ID range. The previous + * version of this file hardcoded `(void *)19`, inside jbpf's built-in range, + * where upstream could later allocate a different function. */ -#define COPY_CHUNK_BYTES 128 -#define MAX_COPY_CHUNKS (MAX_SLOT_IQ_BYTES / COPY_CHUNK_BYTES) +static long (*jbpf_e3_attach)(const struct e3_shm_cfg* cfg, uint64_t cfg_len) = (void*)JBPF_E3_ATTACH_ID; + +static long (*jbpf_e3_publish_slot)( + const void* src, uint64_t src_len, const struct e3_slot_sel* sel, uint64_t sel_len) = + (void*)JBPF_E3_PUBLISH_SLOT_ID; + +/* ---- Maps ---- */ + +/* Descriptors only. Channel name stays "output_map": that is a contract with + * uplink_slot_samples.yaml and slot_iq_pipeline.cpp:75. Only the ELEMENT TYPE + * changes, from the ~733 KB uplink_slot_sample to a ~64 B descriptor — ~64 B each instead of the old 733 KB. */ +jbpf_output_map(output_map, struct e3_slot_desc, 32); + +/* Control input is a CHANNEL: jbpf_control_input_receive() returns 1 only when + * a new message is waiting. So each channel needs a companion array map to hold + * the current value across invocations (eBPF has no writable statics). Same + * pattern as ecpri_iq_collect.c. */ +jbpf_control_input_map(sel_in, struct e3_slot_sel, 1); +jbpf_control_input_map(shm_in, struct e3_shm_cfg, 1); + +struct jbpf_load_map_def SEC("maps") sel_state = { + .type = JBPF_MAP_TYPE_ARRAY, + .key_size = sizeof(int), + .value_size = sizeof(struct e3_slot_sel), + .max_entries = 1, +}; /* ---- Main codelet entry ---- */ -SEC("jbpf_ran_ofh") -uint64_t jbpf_main(void *state) +SEC(JBPF_E3_SEC_RAN_SLOT) +uint64_t +jbpf_main(void* state) { - struct jbpf_ran_slot_ctx *ctx = (struct jbpf_ran_slot_ctx *)state; + struct jbpf_ran_slot_ctx* ctx = (struct jbpf_ran_slot_ctx*)state; + int zero = 0; + + /* FIRST statement: everything after this belongs to the codelet, not to + * jbpf delivery. Paired with the exit stamp further down; see the comment on + * codelet_entry_ts_ns in uplink_slot_data.h for why the two are split. */ + uint64_t entry_ts_ns = jbpf_time_get_ns(); + + /* Region config: forwarded straight to the helper, which attaches lazily. + * The E3Controller owns /e3_ran_buffers; the gNB only ever attaches, so no + * duplicate shm configuration lives on the RAN side. */ + struct e3_shm_cfg cfg; + if (jbpf_control_input_receive(&shm_in, (void*)&cfg, sizeof(cfg)) == 1) { + jbpf_e3_attach(&cfg, sizeof(cfg)); + } - /* Reserve a ring slot directly in jbpf SHM. Writing in-place - * here is the one-and-only memcpy on the codelet side - no temp - * map, no double copy. */ - struct uplink_slot_sample *out = - (struct uplink_slot_sample *)jbpf_get_output_buf(&output_map); + /* Selector: cache on change, read the cache every fire. */ + struct e3_slot_sel incoming; + if (jbpf_control_input_receive(&sel_in, (void*)&incoming, sizeof(incoming)) == 1) { + struct e3_slot_sel* slot = (struct e3_slot_sel*)jbpf_map_lookup_elem(&sel_state, &zero); + if (slot) { + *slot = incoming; + } + } + + struct e3_slot_sel* sel = (struct e3_slot_sel*)jbpf_map_lookup_elem(&sel_state, &zero); + if (!sel) { + return JBPF_CODELET_FAILURE; + } + + /* ---- Temporal filter, BEFORE any data movement ---- + * + * A non-selected slot costs only these comparisons: no helper call, no + * bytes touched, nothing in the PHY RX thread. That is what makes slot + * filtering a real lever on RAN-side cost rather than just a delivery + * filter. + * + * slot_mask == 0 means publish everything — the default, and identical to + * the unfiltered system. */ + uint8_t filtered = 0; + + if (sel->sfn_mod > 1) { + if ((uint16_t)(ctx->sfn % sel->sfn_mod) != sel->sfn_offset) { + return JBPF_CODELET_SUCCESS; + } + filtered = E3_SLOT_FLAG_FILTERED; + } + + if (sel->slot_mask != 0) { + uint16_t sid = ctx->slot_id; + if (sid >= 64) { + /* Outside the mask's width; cannot have been selected. */ + return JBPF_CODELET_SUCCESS; + } + if (((sel->slot_mask >> sid) & 1ULL) == 0) { + return JBPF_CODELET_SUCCESS; + } + filtered = E3_SLOT_FLAG_FILTERED; + } + + /* ---- Publish ---- + * + * Ladder of per-config CONSTANT sizes, largest first. Two properties are + * load-bearing and were established by running the verifier, not by + * reasoning: + * + * 1. The length must be a LITERAL at the call site. Hoisting it into a + * variable and calling once reintroduces a branch join that loses the + * bound proof. + * 2. Each bound check must immediately precede its own call. + * + * The obvious form + * avail = ctx->data_end - ctx->data; clamp; publish(src, avail) + * is REJECTED, even keeping avail in uint64: + * "Upper bound must be at most packet_size + * (valid_access(r1.offset, width=r2) for read)" + * PREVAIL does compute ptr-minus-ptr relationally, but the relation does + * not survive into the ValidAccess assertion. + * + * One rung per supported port count is the cost of accepting 2/4/8 ports + * from a single binary. + */ + const uint8_t* src = (const uint8_t*)(uintptr_t)ctx->data; + const uint8_t* end = (const uint8_t*)(uintptr_t)ctx->data_end; + + long rc = -1; + uint32_t published = 0; + uint8_t flags = filtered; + + if (src + E3_SLOT_BYTES_8P <= end) { + published = E3_SLOT_BYTES_8P; + rc = jbpf_e3_publish_slot(src, E3_SLOT_BYTES_8P, sel, sizeof(*sel)); + } else if (src + E3_SLOT_BYTES_4P <= end) { + published = E3_SLOT_BYTES_4P; + rc = jbpf_e3_publish_slot(src, E3_SLOT_BYTES_4P, sel, sizeof(*sel)); + } else if (src + E3_SLOT_BYTES_2P <= end) { + published = E3_SLOT_BYTES_2P; + rc = jbpf_e3_publish_slot(src, E3_SLOT_BYTES_2P, sel, sizeof(*sel)); + } else { + /* No rung matched: the grid is smaller than the smallest config we know + * how to publish. Emit a descriptor with TRUNCATED set and nothing + * written, so the condition is visible to the controller instead of + * looking like a dead hook. Never publish a prefix — a partial antenna + * set with no signal is a silently wrong measurement. */ + published = 0; + flags |= E3_SLOT_FLAG_TRUNCATED; + } + + if (rc < 0 && (flags & E3_SLOT_FLAG_TRUNCATED) == 0) { + /* Helper refused: not attached, stale epoch, or the selector implied + * more bytes than the row holds. Report rather than retry. */ + published = 0; + flags |= E3_SLOT_FLAG_TRUNCATED; + rc = 0; + } + + /* ---- Descriptor ---- */ + struct e3_slot_desc* out = (struct e3_slot_desc*)jbpf_get_output_buf(&output_map); if (!out) { return JBPF_CODELET_FAILURE; } - /* Two timestamps: - * - gnb_ts_ns : stamped by the ocudu hook caller at hand-off, - * the RAN anchor for RAN -> dApp latency. - * - codelet_ts_ns: captured below, just before jbpf_send_output(), - * so codelet_to_dispatch_us reflects only the - * dispatcher's busy-poll latency rather than the - * codelet's own copy/send cost. The codelet's own - * runtime is then attributed to gnb_to_codelet_us. - * Both use CLOCK_REALTIME ns so they subtract cleanly. */ out->gnb_ts_ns = ctx->gnb_ts_ns; + out->bytes_written = published; + out->fh_buffer_index = (uint8_t)(((uint64_t)rc >> 32) & 0xffULL); + out->fh_write_index = (uint32_t)((uint64_t)rc & 0xffffffffULL); + out->flags = flags; out->sfn = ctx->sfn; out->subframe_id = ctx->subframe_id; out->slot_id = ctx->slot_id; @@ -90,53 +219,12 @@ uint64_t jbpf_main(void *state) out->nof_ports = ctx->nof_ports; out->nof_symbols = ctx->nof_symbols; out->nof_subcarriers = ctx->nof_subcarriers; - out->reserved = 0; - - /* Compute the available payload size from the verifier-tracked - * (ctx->data, ctx->data_end) range, clamp to the static struct - * ceiling. The verifier requires both ends to be visible packet - * pointers; the cast through uintptr is to make the subtraction - * the verifier sees as a plain integer. */ - uint64_t avail = (uint64_t)ctx->data_end - (uint64_t)ctx->data; - if (avail > MAX_SLOT_IQ_BYTES) { - avail = MAX_SLOT_IQ_BYTES; - } - uint32_t copy_size = (uint32_t)avail; - out->iq_size_bytes = copy_size; - - /* Chunked memcpy. The outer loop is bounded by MAX_COPY_CHUNKS at - * compile time; each chunk is a fixed-size memcpy the verifier - * lowers to a small constant instruction sequence. Bounds check - * per chunk against ctx->data_end (required - the verifier - * treats ctx->data as a packet pointer and demands per-access - * bounds) and against copy_size for early termination. */ - const uint8_t *src = (const uint8_t *)(uintptr_t)ctx->data; - const uint8_t *src_end = (const uint8_t *)(uintptr_t)ctx->data_end; - - for (uint32_t i = 0; i < MAX_COPY_CHUNKS; i++) { - uint32_t off = i * COPY_CHUNK_BYTES; - /* Reached the end of the actual payload - leave the rest of - * out->iq untouched (jbpf zeros the ring slot on reserve, so - * untouched bytes are deterministic zeros). */ - if (off >= copy_size) { - break; - } - /* Verifier safety: confirm the next full chunk is in the - * packet-bounded source range before reading. */ - if (src + off + COPY_CHUNK_BYTES > src_end) { - break; - } - __builtin_memcpy(&out->iq[off], src + off, COPY_CHUNK_BYTES); - } - /* Stamp the codelet timestamp here so codelet_to_dispatch_us - * captures only the dispatcher's busy-poll pickup latency, not the - * memcpy + send_output cost above. The memcpy + send-cost shows up - * in gnb_to_codelet_us instead. */ - out->codelet_ts_ns = jbpf_time_get_ns(); + /* Stamped last so the controller's codelet_to_dispatch measurement reflects + * only the dispatcher's pickup latency. */ + out->codelet_entry_ts_ns = entry_ts_ns; + out->codelet_ts_ns = jbpf_time_get_ns(); - /* Submit the slot to the output ring. The E3Controller's - * SlotIqPipeline will see this slot on its next jbpf poll. */ if (jbpf_send_output(&output_map) < 0) { return JBPF_CODELET_FAILURE; } diff --git a/codelets/uplink_slot_samples/uplink_slot_collect.o b/codelets/uplink_slot_samples/uplink_slot_collect.o index 44117d8..eb7a4a8 100644 Binary files a/codelets/uplink_slot_samples/uplink_slot_collect.o and b/codelets/uplink_slot_samples/uplink_slot_collect.o differ diff --git a/codelets/uplink_slot_samples/uplink_slot_data.h b/codelets/uplink_slot_samples/uplink_slot_data.h index 4e7fa1a..4492a4f 100644 --- a/codelets/uplink_slot_samples/uplink_slot_data.h +++ b/codelets/uplink_slot_samples/uplink_slot_data.h @@ -36,17 +36,36 @@ * 4 ports * 14 symbols * 3276 subc * 4 bytes = 733824. */ #define MAX_SLOT_IQ_BYTES 733824 +/* LEGACY. The codelet publishes struct e3_slot_desc (codelets/include/ + * jbpf_e3_slot_api.h), not this. Nothing writes this struct any more; the + * header survives for MAX_SLOT_IQ_BYTES. Kept as the record of the full-IQ + * wire format. + */ struct uplink_slot_sample { - /* gNB-side hand-off timestamp (CLOCK_REALTIME ns) captured by the - * ocudu hook caller, right before the hook fires. This is the RAN - * anchor for end-to-end latency; the dApp subtracts its own - * CLOCK_REALTIME receipt time from it. Same clock domain as - * jbpf_time_get_ns() so codelet_ts_ns below subtracts cleanly. */ + /* A1 entry: the gNB hook caller stamps this immediately before the hook + * fires, on the LAST symbol of the slot, gated on is_valid - so the + * resource grid is complete and nothing has been copied yet. The gNB hands + * the codelet the grid's raw storage pointer and never memcpys it, so this + * is the true "data exists, nothing has moved" instant. + * + * CLOCK_MONOTONIC ns. (The gNB moved off CLOCK_REALTIME; the codelet's + * jbpf_time_get_ns() follows via jbpf_patches/jbpf_monotonic_time.patch, + * and so does the controller's dispatcher poll. All three have to agree or + * the subtractions below silently mix epochs.) */ uint64_t gnb_ts_ns; - /* Codelet entry timestamp (jbpf_time_get_ns(), CLOCK_REALTIME ns). - * Subtract gnb_ts_ns to get the ocudu hook -> codelet entry latency - * (jbpf invocation + verifier path overhead). */ + /* A1 exit / A2 entry, stamped LAST - just before jbpf_send_output(), after + * the codelet has finished moving the slot's data. Same clock as above. + * + * codelet_ts_ns - gnb_ts_ns = A1, the data recording itself. + * + * This is emphatically NOT "hook -> codelet entry latency", and the cost it + * covers is not jbpf plumbing: dispatch is a few microseconds and the data + * movement is tens. Nor is there any verifier cost in it - the gNB loads + * through ubpf, whose JIT emits no bounds checks, and verification is an + * offline build-time gate (see codelets/Makefile). The descriptor path + * splits this into codelet_entry_ts_ns and codelet_ts_ns for exactly that + * reason; this legacy struct keeps the single stamp. */ uint64_t codelet_ts_ns; /* 3GPP slot identification (parsed by the ocudu hook). */ diff --git a/codelets/uplink_slot_samples/uplink_slot_samples.yaml b/codelets/uplink_slot_samples/uplink_slot_samples.yaml index 442cb4a..39d7c3b 100644 --- a/codelets/uplink_slot_samples/uplink_slot_samples.yaml +++ b/codelets/uplink_slot_samples/uplink_slot_samples.yaml @@ -10,3 +10,17 @@ codelet_descriptor: - name: output_map stream_id: "7531abcd1234567890fedcba0987654a" forward_destination: DestinationNone + # Control input. The E3Controller builds the load request programmatically + # (src/e3sm/slot_iq_pipeline.cpp), so these are here for reference and for + # jbpf_lcm_cli users -- keep the names and stream ids in sync with both. + # + # shm_in e3_shm_cfg -- where /e3_ran_buffers is and its row geometry, so + # the gNB-side publish helper can attach. Only sent + # in `writer: gnb` mode. + # sel_in e3_slot_sel -- temporal selector (which slots to publish). + # slot_mask == 0 means publish everything. + in_io_channel: + - name: shm_in + stream_id: "7531abcd1234567890fedcba0987654b" + - name: sel_in + stream_id: "7531abcd1234567890fedcba0987654c" diff --git a/codelets/verifier/CMakeLists.txt b/codelets/verifier/CMakeLists.txt new file mode 100644 index 0000000..e75071f --- /dev/null +++ b/codelets/verifier/CMakeLists.txt @@ -0,0 +1,37 @@ +# e3_verifier_cli — offline PREVAIL verifier for E3Controller's codelets. +# +# Replaces the SDK image's srsran_verifier_cli. Built from the jbpf submodule's +# verifier library, plus the E3 program-type and helper registrations. +# +# Needs ocudu's janus headers for the ctx struct definitions: the context +# descriptors are derived with offsetof() from the RAN's own header rather than +# hard-coded, so the contract has exactly one source of truth. + +if(NOT EXISTS "${OCUDU_JANUS_INC}/jbpf_srsran_contexts.h") + message(WARNING + "e3_verifier_cli: jbpf_srsran_contexts.h not found under OCUDU_JANUS_INC=" + "'${OCUDU_JANUS_INC}'. Skipping the codelet verifier target. Point OCUDU_DIR " + "at an ocudu-e3 checkout to build it:\n" + " cmake -DOCUDU_DIR=/path/to/ocudu-e3 ...") + return() +endif() + +add_executable(e3_verifier_cli e3_verifier_cli.cpp) + +target_link_libraries(e3_verifier_cli PRIVATE jbpf::verifier_lib) + +set(_e3v_jbpf "${CMAKE_CURRENT_SOURCE_DIR}/../../jbpf") +target_include_directories(e3_verifier_cli PRIVATE + ${CMAKE_CURRENT_SOURCE_DIR}/../include # jbpf_e3_ids.h + ${OCUDU_JANUS_INC} # jbpf_srsran_contexts.h (contract) + ${_e3v_jbpf}/src/verifier # jbpf_verifier.hpp + ${_e3v_jbpf}/src/verifier/specs + ${_e3v_jbpf}/src/common # jbpf_defs.h, jbpf_helper_api_defs_ext.h + ${_e3v_jbpf}/3p/ebpf-verifier/src # spec_type_descriptors.hpp, platform.hpp + ${_e3v_jbpf}/3p/ebpf-verifier/src/main + ${_e3v_jbpf}/3p/ebpf-verifier/external +) + +set_target_properties(e3_verifier_cli PROPERTIES + RUNTIME_OUTPUT_DIRECTORY "${OUTPUT_DIR}/bin" +) diff --git a/codelets/verifier/bench_slot_convert.cpp b/codelets/verifier/bench_slot_convert.cpp new file mode 100644 index 0000000..516735c --- /dev/null +++ b/codelets/verifier/bench_slot_convert.cpp @@ -0,0 +1,161 @@ +/* + * bench_slot_convert — microbenchmark for the cbf16 -> fp16 slot convert (B3). + * + * The one piece of workstream B that carries real risk. This loop moves from a + * dedicated E3Controller worker core into the gNB's PHY RX thread, inline in the + * hook, so its cost lands in the RAN's per-slot budget rather than on spare + * controller CPU. + * + * What we need to know before committing: + * 1. Does the fused convert stay memory-bound, i.e. roughly as fast as a plain + * memcpy of the same volume? If yes, replacing + * "codelet copy (56us) + controller convert (74us)" with one fused pass is + * a clear win. If not, the RX thread pays. + * 2. How much does the scalar fallback cost? That quantifies the damage if the + * -mavx2/-mf16c flags ever go missing, which is exactly how the + * adaptive_cpu bimodal-latency bug happened. + * + * Build (must be explicit about the ISA — that is the point): + * g++ -O2 -std=c++17 -mavx2 -mf16c -I../include bench_slot_convert.cpp -o bench_slot_convert + * g++ -O2 -std=c++17 -DE3_ALLOW_SCALAR_CONVERT -DBENCH_FORCE_SCALAR \ + * -I../include bench_slot_convert.cpp -o bench_slot_convert_scalar + */ + +#include +#include +#include +#include +#include +#include +#include +#include + +#ifdef BENCH_FORCE_SCALAR +/* Deliberately hide the vector path so we can price the fallback. */ +#undef __AVX2__ +#undef __F16C__ +#define E3_ALLOW_SCALAR_CONVERT 1 +#endif + +#include "e3_slot_convert.h" + +using clk = std::chrono::steady_clock; + +namespace { + +struct Result { + double us_median; + double us_p95; + double gbps; /* counting bytes read + bytes written */ +}; + +/* Template rather than std::function: an indirect call inside the timing loop + * would be measured along with the work. */ +template +Result +time_it(std::size_t bytes_moved, int iters, Fn&& fn) +{ + std::vector samples; + samples.reserve(iters); + for (int i = 0; i < iters; ++i) { + auto t0 = clk::now(); + fn(); + auto t1 = clk::now(); + samples.push_back(std::chrono::duration(t1 - t0).count()); + } + std::sort(samples.begin(), samples.end()); + Result r{}; + r.us_median = samples[samples.size() / 2]; + r.us_p95 = samples[static_cast(samples.size() * 0.95)]; + r.gbps = static_cast(bytes_moved) / (r.us_median * 1e-6) / 1e9; + return r; +} + +void +run_config(std::uint32_t nof_ports, int iters) +{ + e3_convert::RowGeometry geom{14u, 3276u, nof_ports}; + const std::size_t n_u16_total = geom.row_u16(); + const std::size_t src_bytes = n_u16_total * sizeof(std::uint16_t); + + std::vector src(n_u16_total); + std::vector dst(n_u16_total, 0); + std::vector memcpy_dst(n_u16_total, 0); + + /* bf16 patterns spanning a realistic magnitude range, so the convert hits + * normals rather than a degenerate all-zero fast case. */ + std::mt19937 rng(12345); + std::uniform_int_distribution pick(0x3000u, 0x4100u); + for (auto& s : src) { + s = static_cast(pick(rng)); + } + + const float scale = 1.0f; + + /* touch both buffers once so we are not measuring first-touch page faults */ + std::memset(dst.data(), 0, src_bytes); + std::memset(memcpy_dst.data(), 0, src_bytes); + + const std::size_t traffic = src_bytes * 2; /* read + write */ + + Result conv = time_it(traffic, iters, [&] { + e3_convert::convert_slot(dst.data(), src.data(), geom, nof_ports, scale); + }); + + Result mcpy = time_it(traffic, iters, [&] { std::memcpy(memcpy_dst.data(), src.data(), src_bytes); }); + + std::printf(" %u ports %8zu B/slot convert %7.1f us (p95 %7.1f) %5.1f GB/s" + " | memcpy %7.1f us %5.1f GB/s | ratio %4.2fx\n", + nof_ports, src_bytes, conv.us_median, conv.us_p95, conv.gbps, mcpy.us_median, mcpy.gbps, + conv.us_median / mcpy.us_median); +} + +} // namespace + +int +main(int argc, char** argv) +{ + int iters = (argc > 1) ? std::atoi(argv[1]) : 200; + + std::printf("bench_slot_convert — cbf16 -> fp16, 14 symbols x 3276 subcarriers\n"); + std::printf("vector path compiled in: %s\n", e3_convert::has_vector_path() ? "YES (AVX2+F16C)" : "NO (SCALAR)"); + std::printf("iterations per config: %d (median / p95 reported)\n", iters); + std::printf("throughput counts bytes READ + WRITTEN\n\n"); + + /* Correctness: vector and scalar must agree bit-for-bit, or the two writers + * would produce different rows and acceptance criterion 4 (byte-identical + * dApp input) is meaningless. */ + { + e3_convert::RowGeometry geom{14u, 3276u, 1u}; + const std::size_t n = geom.u16_per_ant(); + std::vector src(n), a(n), b(n); + for (std::size_t i = 0; i < n; ++i) { + src[i] = static_cast((i * 2654435761u) & 0xffffu); + } + e3_convert::convert_ant(a.data(), src.data(), n, 1.0f); + for (std::size_t i = 0; i < n; ++i) { + b[i] = e3_convert::bf16_to_fp16_scalar(src[i], 1.0f); + } + std::size_t mismatches = 0; + for (std::size_t i = 0; i < n; ++i) { + if (a[i] != b[i]) { + ++mismatches; + } + } + std::printf("vector vs scalar agreement over %zu samples: %s (%zu mismatches)\n\n", n, + mismatches == 0 ? "IDENTICAL" : "MISMATCH", mismatches); + if (mismatches) { + return 1; + } + } + + for (std::uint32_t ports : {2u, 4u, 8u}) { + run_config(ports, iters); + } + + std::printf("\nreference points from the deployed system:\n"); + std::printf(" codelet eBPF copy of 733,824 B : ~56 us (gnb_to_codelet, in RX thread)\n"); + std::printf(" controller publish_row_cbf16 : ~74 us (on a pinned worker core)\n"); + std::printf(" fusing them should cost ~ the 4-port convert figure above, in the RX thread.\n"); + return 0; +} diff --git a/codelets/verifier/e3_verifier_cli.cpp b/codelets/verifier/e3_verifier_cli.cpp new file mode 100644 index 0000000..d3a5044 --- /dev/null +++ b/codelets/verifier/e3_verifier_cli.cpp @@ -0,0 +1,334 @@ +/* + * e3_verifier_cli — PREVAIL verifier for E3Controller's codelets. + * + * Replaces the SDK image's `srsran_verifier_cli`, which only models + * `jbpf_ran_ofh_ctx` (27 B) and therefore cannot verify the per-slot codelet + * at all ("Upper bound must be at most 27"). + * + * WHY THIS MATTERS: the gNB never verifies. jbpf's load path is + * ubpf_load_elf_ex + ubpf_compile (external/jbpf/src/core/jbpf.c:968,976), + * PREVAIL is not on it, and ubpf's JIT emits no memory bounds checks at all + * (only the interpreter has them, 3p/ubpf/vm/ubpf_vm.c:1724). So this binary, + * run at build time, is the ONLY memory-safety gate that will ever execute on + * these codelets. codelets/Makefile.common treats its exit status as fatal. + * + * Registers, on top of jbpf's built-in program/helper/map tables: + * - program type jbpf_ran_ofh (ecpri_iq_samples) + * - program type jbpf_ran_slot (uplink_slot_samples) + * - helper prototypes for the two E3 helpers in jbpf_e3_ids.h + * + * Usage: e3_verifier_cli [section] [--asm ] + * Exit: 0 = verified, 1 = failed (matching jbpf_verifier_cli's `return !res`). + */ + +#include +#include +#include +#include +#include +#include +#include + +#include "jbpf_verifier.hpp" +#include "spec_type_descriptors.hpp" + +extern "C" { +#include "jbpf_defs.h" +#include "jbpf_helper_api_defs_ext.h" +#include "jbpf_srsran_contexts.h" /* imported from ocudu — see Makefile.defs */ +} + +#include "jbpf_e3_ids.h" + +/* ------------------------------------------------------------------ * + * Context descriptors, derived from the ocudu contract header + * ------------------------------------------------------------------ */ + +/* + * CAUTION: PREVAIL's context `size` is the UNPADDED extent — the offset of the + * last field plus its size — NOT sizeof(). Getting this wrong silently changes + * what the verifier considers an in-bounds ctx read. + * + * Evidence, both directions: + * - jbpf's own extension test uses 31 for a struct whose sizeof() is 32 + * (jbpf_tests/verifier/jbpf_verifier_extension_test.cpp). + * - the error we currently get for jbpf_ran_ofh_ctx says "at most 27", + * while sizeof(struct jbpf_ran_ofh_ctx) is 32. + * + * So: offsetof(last field) + sizeof(last field). Deriving it instead of + * hard-coding it means a field added on the RAN side breaks this build rather + * than silently verifying the codelet against a stale layout. + */ +#define CTX_EXTENT(type, last_field) \ + (static_cast(offsetof(type, last_field) + sizeof(static_cast(nullptr)->last_field))) + +/* + * The `meta` slot is deliberately -1 ("no metadata region") for BOTH contexts, + * even though each struct has a field literally named `meta_data`. + * + * PREVAIL's `meta` means an XDP-style packet-metadata POINTER: it models reads + * at that offset as a pointer type, and -1 is its documented absent sentinel + * (`crab/ebpf_domain.cpp:1968` guards on `meta >= 0`; + * `ebpf_yaml.cpp:217` uses `{64, 0, 8, -1}`). + * + * Neither RAN hook supplies packet metadata. Worse, on the eCPRI path + * `meta_data` is explicitly REPURPOSED to carry the gNB's CLOCK_REALTIME + * hand-off timestamp — see the comment on the field in ocudu's + * jbpf_srsran_contexts.h. Declaring it as `meta` makes the verifier type that + * read as a pointer, so the codelet's perfectly legitimate + * `out->timestamp = ctx->meta_data` is rejected with: + * + * Only numbers can be stored to externally-visible regions + * (r0.type != stack -> r2.type == number) + * + * Observed exactly that. With -1 the read is a plain scalar, which is what the + * field actually holds. + * + * Note also what that field comment says: meta_data was reused *because* + * "extending the struct would require rebuilding srsran_verifier_cli from + * source". This tool removes that constraint — so adding a proper `gnb_ts_ns` + * field to jbpf_ran_ofh_ctx (as jbpf_ran_slot_ctx already has) is now possible + * and would be cleaner than the reuse. Left as a follow-up: it is a RAN-side + * contract change and invalidates every compiled codelet. + */ +static constexpr int kNoMetaRegion = -1; + +static constexpr ebpf_context_descriptor_t g_ran_ofh_descr = { + CTX_EXTENT(struct jbpf_ran_ofh_ctx, direction), + static_cast(offsetof(struct jbpf_ran_ofh_ctx, data)), + static_cast(offsetof(struct jbpf_ran_ofh_ctx, data_end)), + kNoMetaRegion, +}; + +static constexpr ebpf_context_descriptor_t g_ran_slot_descr = { + CTX_EXTENT(struct jbpf_ran_slot_ctx, direction), + static_cast(offsetof(struct jbpf_ran_slot_ctx, data)), + static_cast(offsetof(struct jbpf_ran_slot_ctx, data_end)), + kNoMetaRegion, +}; + +/* Tripwires: if either fires, the RAN-side ctx layout changed. Re-derive the + * expectation deliberately rather than just bumping the number — a moved + * data/data_end/meta_data offset also invalidates every compiled codelet. */ +static_assert(g_ran_ofh_descr.size == 27, "jbpf_ran_ofh_ctx layout changed (expected 27 B extent)"); +static_assert(g_ran_slot_descr.size == 47, "jbpf_ran_slot_ctx layout changed (expected 47 B extent)"); +static_assert(g_ran_slot_descr.data == 0 && g_ran_slot_descr.end == 8, "jbpf_ran_slot_ctx data/data_end moved"); +static_assert(g_ran_ofh_descr.data == 0 && g_ran_ofh_descr.end == 8, "jbpf_ran_ofh_ctx data/data_end moved"); + +/* ------------------------------------------------------------------ * + * Helper prototypes + * ------------------------------------------------------------------ */ + +/* + * Field order is {name, return_type, argument_type[5], reallocate_packet, + * context_descriptor} — see 3p/ebpf-verifier/src/helpers.hpp. Positional + * initialisation (rather than designators) keeps this compiling under a strict + * -std=c++17, which is what E3Controller sets. + * + * context_descriptor is left null so both helpers are callable from any + * program type; jbpf_verifier_is_helper_usable() only enforces a match when + * it is non-null. + */ + +/* jbpf_e3_attach(cfg, cfg_len) */ +static const EbpfHelperPrototype g_e3_attach_proto = { + "e3_attach", + EBPF_RETURN_TYPE_INTEGER, + { + EBPF_ARGUMENT_TYPE_PTR_TO_READABLE_MEM, + EBPF_ARGUMENT_TYPE_CONST_SIZE_OR_ZERO, + EBPF_ARGUMENT_TYPE_DONTCARE, + EBPF_ARGUMENT_TYPE_DONTCARE, + EBPF_ARGUMENT_TYPE_DONTCARE, + }, + false, + nullptr, +}; + +/* + * jbpf_e3_publish_slot(src, src_len, sel, sel_len) + * + * Only the SOURCE needs verifier-proven bounds. The destination is not an + * argument at all — the helper derives the row address from geometry it owns, + * so a codelet cannot forge it. + * + * The (ptr, size) pairing is positional and REQUIRED: PREVAIL only forms a + * checked pair when the size register immediately follows the pointer register + * (3p/ebpf-verifier/src/asm_unmarshal.cpp:337-343). This is why the old + * jbpf_native_memcpy(dst, src, len) signature could never be verified, even + * with a prototype registered. + */ +static const EbpfHelperPrototype g_e3_publish_slot_proto = { + "e3_publish_slot", + EBPF_RETURN_TYPE_INTEGER, + { + EBPF_ARGUMENT_TYPE_PTR_TO_READABLE_MEM, + EBPF_ARGUMENT_TYPE_CONST_SIZE_OR_ZERO, + EBPF_ARGUMENT_TYPE_PTR_TO_READABLE_MEM, + EBPF_ARGUMENT_TYPE_CONST_SIZE_OR_ZERO, + EBPF_ARGUMENT_TYPE_DONTCARE, + }, + false, + nullptr, +}; + +/* ------------------------------------------------------------------ * + * Registration + * ------------------------------------------------------------------ */ + +/* + * Poison entry for helper IDs that nothing has registered. + * + * WHY THIS IS NEEDED — jbpf bug, worth reporting upstream: + * + * jbpf_verifier_register_helper() grows its table with + * `prototypes.resize(helper_id + 1)` (specs/jbpf_prototypes.cpp:271). Registering + * anything at CUSTOM_HELPER_START_ID (32) therefore leaves IDs 19..31 + * default-constructed, i.e. zeroed — and a zeroed entry has + * context_descriptor == nullptr, which is precisely what + * jbpf_verifier_is_helper_usable() treats as "usable from any program type". + * + * So a codelet calling an unregistered helper in that gap gets past the + * usability check, reaches `res.name = proto.name` in + * 3p/ebpf-verifier/src/asm_unmarshal.cpp:310 with proto.name == nullptr, and + * constructs a std::string from a null pointer: SIGSEGV, no diagnostic. + * + * Observed exactly that on the current uplink_slot_collect.o, which still calls + * the old jbpf_native_memcpy stub at ID 19. + * + * EBPF_RETURN_TYPE_UNSUPPORTED is handled properly by makeCall(), which throws + * `Unsupported function: ` — so filling the gap turns a crash into a + * message that names the offending helper. + */ +static EbpfHelperPrototype +make_unregistered_proto(int id) +{ + /* Names must outlive registration (EbpfHelperPrototype::name is a raw + * const char*), and we want the ID in the message so the diagnostic points + * at the actual call. Deque-like stable storage: a static list that is only + * ever appended to before use, never reallocated afterwards. */ + static std::vector> names; + names.push_back(std::make_unique("")); + + EbpfHelperPrototype p{}; + p.name = names.back()->c_str(); + p.return_type = EBPF_RETURN_TYPE_UNSUPPORTED; + p.argument_type[0] = EBPF_ARGUMENT_TYPE_DONTCARE; + p.argument_type[1] = EBPF_ARGUMENT_TYPE_DONTCARE; + p.argument_type[2] = EBPF_ARGUMENT_TYPE_DONTCARE; + p.argument_type[3] = EBPF_ARGUMENT_TYPE_DONTCARE; + p.argument_type[4] = EBPF_ARGUMENT_TYPE_DONTCARE; + p.reallocate_packet = false; + p.context_descriptor = nullptr; + return p; +} + +static void +register_e3_extensions() +{ + /* Program types. Lookup is by ELF section prefix, first match wins + * (jbpf_platform.cpp:45-58). "jbpf_ran_ofh" and "jbpf_ran_slot" are not + * prefixes of one another, so registration order is not load-bearing. */ + EbpfProgramType ran_ofh; + ran_ofh.name = "jbpf_ran_ofh"; + ran_ofh.context_descriptor = &g_ran_ofh_descr; + ran_ofh.platform_specific_data = JBPF_PROG_TYPE_RAN_OFH; + ran_ofh.section_prefixes = {JBPF_E3_SEC_RAN_OFH}; + ran_ofh.is_privileged = false; + jbpf_verifier_register_program_type(JBPF_PROG_TYPE_RAN_OFH, ran_ofh); + + EbpfProgramType ran_slot; + ran_slot.name = "jbpf_ran_slot"; + ran_slot.context_descriptor = &g_ran_slot_descr; + ran_slot.platform_specific_data = JBPF_E3_PROG_TYPE_RAN_SLOT; + ran_slot.section_prefixes = {JBPF_E3_SEC_RAN_SLOT}; + ran_slot.is_privileged = false; + jbpf_verifier_register_program_type(JBPF_E3_PROG_TYPE_RAN_SLOT, ran_slot); + + /* Close the gap between jbpf's built-ins and our custom range BEFORE + * registering ours, so no zeroed entry is ever reachable. See the comment + * on g_unregistered_proto. JBPF_NUM_HELPERS_MAX is one past the last + * built-in, so it is the correct lower bound whichever jbpf we build + * against. */ + static_assert(JBPF_NUM_HELPERS_MAX <= JBPF_E3_ATTACH_ID, + "jbpf built-in helper range now overlaps the E3 custom range"); + for (int id = JBPF_NUM_HELPERS_MAX; id < JBPF_E3_ATTACH_ID; ++id) { + jbpf_verifier_register_helper(id, make_unregistered_proto(id)); + } + + /* Helpers. */ + jbpf_verifier_register_helper(JBPF_E3_ATTACH_ID, g_e3_attach_proto); + jbpf_verifier_register_helper(JBPF_E3_PUBLISH_SLOT_ID, g_e3_publish_slot_proto); +} + +/* ------------------------------------------------------------------ */ + +static void +usage(const char* argv0) +{ + std::fprintf(stderr, + "usage: %s [section] [--asm ]\n" + "\n" + " Verifies a codelet against the E3 program types and helpers.\n" + " Exit status 0 means verified; 1 means it did not verify.\n", + argv0); +} + +int +main(int argc, char** argv) +{ + const char* path = nullptr; + const char* section = nullptr; + const char* asmfile = nullptr; + + for (int i = 1; i < argc; ++i) { + if (std::strcmp(argv[i], "--asm") == 0) { + if (++i >= argc) { + usage(argv[0]); + return 1; + } + asmfile = argv[i]; + } else if (std::strcmp(argv[i], "-h") == 0 || std::strcmp(argv[i], "--help") == 0) { + usage(argv[0]); + return 0; + } else if (path == nullptr) { + path = argv[i]; + } else if (section == nullptr) { + section = argv[i]; + } else { + usage(argv[0]); + return 1; + } + } + + if (path == nullptr) { + usage(argv[0]); + return 1; + } + + register_e3_extensions(); + + /* jbpf_verify() lets exceptions from the unmarshaller escape (e.g. an + * unregistered helper ID, or a malformed object). Catch them so the tool + * reports a diagnosis and exit status 1 rather than std::terminate. */ + jbpf_verifier_result_t result{}; + try { + result = jbpf_verify(path, section, asmfile); + } catch (const std::exception& e) { + std::fprintf(stderr, "FAILED verification: %s: %s\n", path, e.what()); + return 1; + } catch (...) { + std::fprintf(stderr, "FAILED verification: %s: unknown error\n", path); + return 1; + } + + if (!result.verification_pass) { + std::fprintf(stderr, "FAILED verification: %s: %s\n", path, result.err_msg); + return 1; + } + + std::printf( + "verified: %s (%.3fs, terminates within %lu instructions)\n", path, result.runtime_seconds, + result.max_instruction_count); + return 0; +} diff --git a/configs/e3_controller.yaml b/configs/e3_controller.yaml new file mode 100644 index 0000000..0848d78 --- /dev/null +++ b/configs/e3_controller.yaml @@ -0,0 +1,206 @@ +# E3Controller configuration. +# +# This is the only argument the controller takes: e3_controller --config +# +# Unknown keys are a startup ERROR, not a warning. A typo'd `nof_port:` that was +# silently ignored would leave the controller sizing rows for the default while +# you believed you had configured something else — which is the exact failure +# this file exists to prevent. + +# --------------------------------------------------------------------------- +# Radio geometry — MUST match the running gNB. +# +# Checked, not trusted: the controller compares these against the geometry the +# RAN reports in the first slot (jbpf_ran_slot_ctx carries nof_ports / +# nof_symbols / nof_subcarriers) and refuses to publish on a mismatch. +# +# The asymmetry worth knowing: declaring FEWER ports than the gNB sends is +# already safe (the gNB-side helper refuses and the codelet flags it). Declaring +# MORE is what needs the check — the surplus antennas get written as silence and +# the dApp cannot tell them from a genuinely quiet antenna. +# --------------------------------------------------------------------------- +radio: + nof_ports: 4 # UL antenna ports (1..8) + nof_prbs: 273 # 100 MHz @ 30 kHz SCS + nof_symbols: 14 # per slot (12 or 14) + scs_khz: 30 # 15 | 30 | 60 | 120 — also fixes slots/frame (20 @ 30 kHz) + +# --------------------------------------------------------------------------- +# /e3_ran_buffers — the region the controller OWNS and the dApp reads. +# Layout matches SharedMemoryHeader in the NVIDIA L1 KPM SM / cuBB contract. +# --------------------------------------------------------------------------- +shm: + name: /e3_ran_buffers + size_bytes: 176117824 # 168 MiB = 64 B header + 2 x 120 x 733,824 B + # Chosen to MATCH NVIDIA Aerial's default FH ring depth, so our geometry is + # like-for-like with a cuBB deployment rather than an arbitrary local choice. + # + # Aerial does not define the segment size as a constant. cuBB's E3 agent takes + # the depth as a constructor parameter, defaulted in + # aerial-cuda-accelerated-ran/cuPHY-CP/data_lake/data_lake.hpp: + # + # numRowsToInsertFh = 120 <-- this is what we match + # numRowsToInsertPusch = 400 + # numRowsToInsertHest = 200 + # + # and publishes the result through SharedMemoryHeader::num_fh_rows, whose + # layout in cuPHY-CP/data_lake/e3_agent.hpp is field-for-field identical to + # ours in src/e3sm/utils/e3sm_shm_writer.h. The key string matches too: + # E3_SHARED_MEMORY_KEY = "/e3_ran_buffers". + # + # Sizing, from e3sm_shm_writer.cpp:176-187 — the region is a DOUBLE-buffered + # ring, so rows-per-buffer = (size_bytes - sizeof(SharedMemoryHeader)) / 2 / + # row_bytes, and row_bytes for 4 ports x 273 PRB x 14 sym is 733,824 B. + # + # 1 GiB -> 731 rows/buffer (1462 total, ~2.2 s at 650 slots/s) + # 168 MiB -> 120 rows/buffer ( 240 total, ~370 ms) <-- here, = Aerial + # 128 MiB -> 91 rows/buffer ( 182 total, ~280 ms) + # 24 MiB -> 17 rows/buffer ( 34 total, ~52 ms) + # + # None of this is a latency lever: the dApp drains a row in ~150 us end to end, + # so even 24 MiB is ~340x more buffering than the consumer needs. Depth only + # decides how long a stalled consumer can fall behind before rows are + # overwritten. What 1 GiB did cost was a 1 GiB zero-fill at open() (the memset + # on line 164) and a quarter of this pod's 4 GiB /dev/shm, which also holds the + # 1 GiB jbpf IO region below. 168 MiB + 1 GiB still fits comfortably. + # + # We allocate FH only: pusch_buffer_size and hest_buffer_size are set to 0 + # (e3sm_shm_writer.cpp:191-196), so unlike cuBB there are no PUSCH/HEST regions + # after the FH pair. + # + # Do NOT hardcode the row count anywhere downstream — read num_fh_rows from the + # header. The dApp's RX-link drop accounting used to assume 16 rows, which was + # wrong here AND wrong against Aerial's 120, and manufactured a steady ~17% + # phantom loss. + # + # Too small is loud, not silent: under one row per buffer, open() fails with + # "SHM size N too small for one row". + + # bf16 -> fp16 scale, applied when converting the resource grid to the row. + # Was the E3_CBF16_SCALE environment variable; it lives here now because the + # gNB-side publish helper needs the SAME value (pushed down via e3_shm_cfg), + # and a disagreement would produce different rows with no error at all — + # bf16 and fp16 are both 2 bytes, so a wrong scale is silently wrong data. + # + # Must be > 0. Zero or negative zeroes the row and breaks the dApp silently. + cbf16_scale: 1.0 + + # Which process converts the grid and writes the fp16 rows. + # + # EXACTLY ONE may write. Both this controller and the gNB-side publish helper + # keep their own ring cursor, so if both were active they would overwrite each + # other's rows with no error at all. This switch makes that impossible rather + # than a convention: in `gnb` mode the controller's writer hard-refuses. + # + # controller today's path -- the codelet copies the grid into the jbpf ring + # and this process converts it. Works against any gNB. + # + # gnb publish-direct -- the codelet calls the gNB-side helper, which + # converts and writes the row in the PHY RX thread. The jbpf ring + # then carries only a ~64 B descriptor and this process never + # touches the data plane. Halves total memory traffic, but + # REQUIRES a gNB with the E3 helpers registered and the + # descriptor-emitting codelet loaded. + # + # Either way the controller OWNS the region: it creates, sizes, zero-fills, + # headers and tears it down. The helper only attaches. + writer: gnb + +# --------------------------------------------------------------------------- +# jbpf IPC. Must agree with the gNB's own `jbpf:` YAML section. +# --------------------------------------------------------------------------- +jbpf: + ipc_name: e3_controller + run_path: /dev/shm + mem_size_bytes: 1073741824 + # LCM socket the gNB exposes for codelet loading. + # + # The gNB builds this path as // + # from its own YAML, so with the usual + # jbpf_run_path: "/dev/shm", jbpf_namespace: "jbpf", jbpf_lcm_ipc_name: "jbpf_lcm_ipc" + # it lands at /dev/shm/jbpf/jbpf_lcm_ipc — NOT /tmp. + # + # Leave this commented out and it is derived from run_path above, which keeps + # the two in step automatically. Set it only if your gNB uses a non-default + # namespace or socket name. + # lcm_socket_path: /dev/shm/jbpf/jbpf_lcm_ipc + # Directory holding the codeletsets. Resolved by the LCM load request, so it + # must be the path as seen by the gNB. Empty disables auto-loading. + codelet_base_path: /workspace/E3Controller/codelets + +# --------------------------------------------------------------------------- +# E3 channel. One encoding is served at a time; dApps must speak the same one. +# ASN.1 (APER) is the O-RAN default; JSON matches the cuBB convention the +# adaptive_cpu dApp uses (with ports 5555/5556/5557). +# --------------------------------------------------------------------------- +e3: + # JSON, because the adaptive_cpu dApp on feat/latency-tracing is JSON-ONLY: + # it ships no asn1_receiver and parses no --encoding flag, and its + # config/e3_config.json declares the NVIDIA_L1 agent on 5555/5556/5557. ASN.1 + # here would leave the two talking past each other on different ports. + # + # (The dApp's `feat/asn` branch does speak APER on 9990/9991/9999, but it + # deliberately excludes the latency instrumentation, so it is not the branch + # to use for a tracing run.) + encoding: json # asn1 | json + link_layer: zmq # zmq | posix + transport: tcp # tcp | ipc | sctp + # Must match e3_manager.e3_agents[NVIDIA_L1] in the dApp's e3_config.json. + setup_port: 5555 # agent_rep_port + publisher_port: 5556 # agent_pub_port + subscriber_port: 5557 # agent_sub_port + +# --------------------------------------------------------------------------- +# Thread placement. -1 = no pinning (the poll thread then sleeps between polls +# rather than busy-spinning). +# +# Pin these. Unpinned is not a neutral default here: it cost zmq_send 2.66 -> +# 16.35 us and L2->L3 send 5.44 -> 19.52 us in back-to-back runs, because the +# publisher loses socket/cache locality. It does NOT help the outbound queue +# wait (53.32 vs 52.31 us either way) -- that one is set by the poller's period +# in libe3's lockfree_queue.hpp, not by placement. +# +# Cores must exist in the pod's cpuset or pinning fails outright; check with +# grep Cpus_allowed_list /proc/self/status +# Values below are for ocudu-e3-rusim, whose cpuset is the ODD cores 9..31. +# Keep them disjoint from the dApp's, which are set separately in +# run_scripts/start_adaptive_cpu.sh (CORES=). The srsran-janus-dpdk-x86 pod had +# the EVEN map and wanted 38/40/42 instead. +# --------------------------------------------------------------------------- +threads: + poll_core: 28 + worker_core: 30 + publisher_core: 32 + poll_interval_us: 100 + +logging: + # Throttled drop-accounting CSV. Empty disables it. + # + # One cumulative row per second: uptime, published, dropped_total, + # latrec_clamped, and a column per drop reason (no_subscribers, + # ran_published_nothing, blob_too_small, ...). The drop COUNTERS themselves are + # always on: the throttled live stderr line and the shutdown summary do not + # depend on this path. + # + # This is the only file the L1-KPM SM writes. The per-slot stage CSV that used + # to live here is gone: it opened an ofstream, wrote a row and flushed it once + # per slot inside the sample handler, so the numbers it produced included the + # cost of producing them. Stage timing is latrec's job now. + drops_log_path: "/tmp/e3c_drops.csv" + + # Per-slot stage CSV for the LEGACY eCPRI SM (RF=1) only. Empty disables it. + spectrum_stats_log_path: "" + + # Where latrec writes its per-thread stage-record rings. + # + # PLACEMENT ONLY, not a switch. Whether anything is recorded at all is decided + # when libe3 is built, by -DLIBE3_ENABLE_LATREC (./build.sh --latrec); a normal + # build has no recorder in the process and ignores this. Empty leaves libe3's + # compiled-in default (/tmp/latrec unless the build overrode it). + # + # Rings default to 2^18 records, 8 MiB per thread. The slot path writes five + # records per published slot, so at 2000 slots/s that is ~26 s before the ring + # wraps -- and a wrapped capture is not a valid measurement. Raise it with + # LATREC_ENTRIES_LOG2_E3CONTROLLER_L1_KPM=24 (512 MiB, ~28 min); ceiling 2^28. + latrec_dir: "" +target_slot: -1 diff --git a/configs/e3_controller_asn1.yaml b/configs/e3_controller_asn1.yaml new file mode 100644 index 0000000..09ae856 --- /dev/null +++ b/configs/e3_controller_asn1.yaml @@ -0,0 +1,218 @@ +# E3Controller configuration -- ASN.1 (APER) variant. +# +# Start with: e3_controller --config configs/e3_controller_asn1.yaml +# and the dApp with: ./start_dapp.sh --encoding asn1 --cores ... +# +# Differs from e3_controller.yaml ONLY in the e3: block (encoding + the +# three ports). Everything else is identical on purpose. +# +# This is the only argument the controller takes: e3_controller --config +# +# Unknown keys are a startup ERROR, not a warning. A typo'd `nof_port:` that was +# silently ignored would leave the controller sizing rows for the default while +# you believed you had configured something else — which is the exact failure +# this file exists to prevent. + +# --------------------------------------------------------------------------- +# Radio geometry — MUST match the running gNB. +# +# Checked, not trusted: the controller compares these against the geometry the +# RAN reports in the first slot (jbpf_ran_slot_ctx carries nof_ports / +# nof_symbols / nof_subcarriers) and refuses to publish on a mismatch. +# +# The asymmetry worth knowing: declaring FEWER ports than the gNB sends is +# already safe (the gNB-side helper refuses and the codelet flags it). Declaring +# MORE is what needs the check — the surplus antennas get written as silence and +# the dApp cannot tell them from a genuinely quiet antenna. +# --------------------------------------------------------------------------- +radio: + nof_ports: 4 # UL antenna ports (1..8) + nof_prbs: 273 # 100 MHz @ 30 kHz SCS + nof_symbols: 14 # per slot (12 or 14) + scs_khz: 30 # 15 | 30 | 60 | 120 — also fixes slots/frame (20 @ 30 kHz) + +# --------------------------------------------------------------------------- +# /e3_ran_buffers — the region the controller OWNS and the dApp reads. +# Layout matches SharedMemoryHeader in the NVIDIA L1 KPM SM / cuBB contract. +# --------------------------------------------------------------------------- +shm: + name: /e3_ran_buffers + size_bytes: 176117824 # 168 MiB = 64 B header + 2 x 120 x 733,824 B + # Chosen to MATCH NVIDIA Aerial's default FH ring depth, so our geometry is + # like-for-like with a cuBB deployment rather than an arbitrary local choice. + # + # Aerial does not define the segment size as a constant. cuBB's E3 agent takes + # the depth as a constructor parameter, defaulted in + # aerial-cuda-accelerated-ran/cuPHY-CP/data_lake/data_lake.hpp: + # + # numRowsToInsertFh = 120 <-- this is what we match + # numRowsToInsertPusch = 400 + # numRowsToInsertHest = 200 + # + # and publishes the result through SharedMemoryHeader::num_fh_rows, whose + # layout in cuPHY-CP/data_lake/e3_agent.hpp is field-for-field identical to + # ours in src/e3sm/utils/e3sm_shm_writer.h. The key string matches too: + # E3_SHARED_MEMORY_KEY = "/e3_ran_buffers". + # + # Sizing, from e3sm_shm_writer.cpp:176-187 — the region is a DOUBLE-buffered + # ring, so rows-per-buffer = (size_bytes - sizeof(SharedMemoryHeader)) / 2 / + # row_bytes, and row_bytes for 4 ports x 273 PRB x 14 sym is 733,824 B. + # + # 1 GiB -> 731 rows/buffer (1462 total, ~2.2 s at 650 slots/s) + # 168 MiB -> 120 rows/buffer ( 240 total, ~370 ms) <-- here, = Aerial + # 128 MiB -> 91 rows/buffer ( 182 total, ~280 ms) + # 24 MiB -> 17 rows/buffer ( 34 total, ~52 ms) + # + # None of this is a latency lever: the dApp drains a row in ~150 us end to end, + # so even 24 MiB is ~340x more buffering than the consumer needs. Depth only + # decides how long a stalled consumer can fall behind before rows are + # overwritten. What 1 GiB did cost was a 1 GiB zero-fill at open() (the memset + # on line 164) and a quarter of this pod's 4 GiB /dev/shm, which also holds the + # 1 GiB jbpf IO region below. 168 MiB + 1 GiB still fits comfortably. + # + # We allocate FH only: pusch_buffer_size and hest_buffer_size are set to 0 + # (e3sm_shm_writer.cpp:191-196), so unlike cuBB there are no PUSCH/HEST regions + # after the FH pair. + # + # Do NOT hardcode the row count anywhere downstream — read num_fh_rows from the + # header. The dApp's RX-link drop accounting used to assume 16 rows, which was + # wrong here AND wrong against Aerial's 120, and manufactured a steady ~17% + # phantom loss. + # + # Too small is loud, not silent: under one row per buffer, open() fails with + # "SHM size N too small for one row". + + # bf16 -> fp16 scale, applied when converting the resource grid to the row. + # Was the E3_CBF16_SCALE environment variable; it lives here now because the + # gNB-side publish helper needs the SAME value (pushed down via e3_shm_cfg), + # and a disagreement would produce different rows with no error at all — + # bf16 and fp16 are both 2 bytes, so a wrong scale is silently wrong data. + # + # Must be > 0. Zero or negative zeroes the row and breaks the dApp silently. + cbf16_scale: 1.0 + + # Which process converts the grid and writes the fp16 rows. + # + # EXACTLY ONE may write. Both this controller and the gNB-side publish helper + # keep their own ring cursor, so if both were active they would overwrite each + # other's rows with no error at all. This switch makes that impossible rather + # than a convention: in `gnb` mode the controller's writer hard-refuses. + # + # controller today's path -- the codelet copies the grid into the jbpf ring + # and this process converts it. Works against any gNB. + # + # gnb publish-direct -- the codelet calls the gNB-side helper, which + # converts and writes the row in the PHY RX thread. The jbpf ring + # then carries only a ~64 B descriptor and this process never + # touches the data plane. Halves total memory traffic, but + # REQUIRES a gNB with the E3 helpers registered and the + # descriptor-emitting codelet loaded. + # + # Either way the controller OWNS the region: it creates, sizes, zero-fills, + # headers and tears it down. The helper only attaches. + writer: gnb + +# --------------------------------------------------------------------------- +# jbpf IPC. Must agree with the gNB's own `jbpf:` YAML section. +# --------------------------------------------------------------------------- +jbpf: + ipc_name: e3_controller + run_path: /dev/shm + mem_size_bytes: 1073741824 + # LCM socket the gNB exposes for codelet loading. + # + # The gNB builds this path as // + # from its own YAML, so with the usual + # jbpf_run_path: "/dev/shm", jbpf_namespace: "jbpf", jbpf_lcm_ipc_name: "jbpf_lcm_ipc" + # it lands at /dev/shm/jbpf/jbpf_lcm_ipc — NOT /tmp. + # + # Leave this commented out and it is derived from run_path above, which keeps + # the two in step automatically. Set it only if your gNB uses a non-default + # namespace or socket name. + # lcm_socket_path: /dev/shm/jbpf/jbpf_lcm_ipc + # Directory holding the codeletsets. Resolved by the LCM load request, so it + # must be the path as seen by the gNB. Empty disables auto-loading. + codelet_base_path: /workspace/E3Controller/codelets + +# --------------------------------------------------------------------------- +# E3 channel. One encoding is served at a time; dApps must speak the same one. +# ASN.1 (APER) is the O-RAN default; JSON matches the cuBB convention the +# adaptive_cpu dApp uses (with ports 5555/5556/5557). +# --------------------------------------------------------------------------- +e3: + # ASN.1 (APER). This file is the sibling of e3_controller.yaml, which serves + # JSON; keep every other key in the two identical so a json-vs-asn1 comparison + # measures the encoding and nothing else. + # + # The three ports below are NOT free parameters. The dApp's ASN.1 path builds a + # libe3 E3Agent in DAPP role and takes libe3's E3Config DEFAULTS + # (/usr/local/include/libe3/types.hpp:373-375): + # + # setup_port{9990} subscriber_port{9999} publisher_port{9991} + # + # They are not read from the dApp's config/e3_config.json -- that file's ports + # (5555/5556/5557) belong to the JSON E3Manager path only. Change these and the + # dApp will not find the controller, with no error on either side: they simply + # sit on different ports and no indication ever arrives. + encoding: asn1 # asn1 | json + link_layer: zmq # zmq | posix + transport: tcp # tcp | ipc | sctp + # Must match libe3's E3Config defaults -- see the note above. + setup_port: 9990 # libe3 E3Config default + publisher_port: 9991 # libe3 E3Config default + subscriber_port: 9999 # libe3 E3Config default + +# --------------------------------------------------------------------------- +# Thread placement. -1 = no pinning (the poll thread then sleeps between polls +# rather than busy-spinning). +# +# Pin these. Unpinned is not a neutral default here: it cost zmq_send 2.66 -> +# 16.35 us and L2->L3 send 5.44 -> 19.52 us in back-to-back runs, because the +# publisher loses socket/cache locality. It does NOT help the outbound queue +# wait (53.32 vs 52.31 us either way) -- that one is set by the poller's period +# in libe3's lockfree_queue.hpp, not by placement. +# +# Cores must exist in the pod's cpuset or pinning fails outright; check with +# grep Cpus_allowed_list /proc/self/status +# Values below are for ocudu-e3-rusim, whose cpuset is the ODD cores 9..31. +# Keep them disjoint from the dApp's, which are set separately in +# run_scripts/start_adaptive_cpu.sh (CORES=). The srsran-janus-dpdk-x86 pod had +# the EVEN map and wanted 38/40/42 instead. +# --------------------------------------------------------------------------- +threads: + poll_core: 28 + worker_core: 30 + publisher_core: 32 + poll_interval_us: 100 + +logging: + # Throttled drop-accounting CSV. Empty disables it. + # + # One cumulative row per second: uptime, published, dropped_total, + # latrec_clamped, and a column per drop reason (no_subscribers, + # ran_published_nothing, blob_too_small, ...). The drop COUNTERS themselves are + # always on: the throttled live stderr line and the shutdown summary do not + # depend on this path. + # + # This is the only file the L1-KPM SM writes. The per-slot stage CSV that used + # to live here is gone: it opened an ofstream, wrote a row and flushed it once + # per slot inside the sample handler, so the numbers it produced included the + # cost of producing them. Stage timing is latrec's job now. + drops_log_path: "/tmp/e3c_drops.csv" + + # Per-slot stage CSV for the LEGACY eCPRI SM (RF=1) only. Empty disables it. + spectrum_stats_log_path: "" + + # Where latrec writes its per-thread stage-record rings. + # + # PLACEMENT ONLY, not a switch. Whether anything is recorded at all is decided + # when libe3 is built, by -DLIBE3_ENABLE_LATREC (./build.sh --latrec); a normal + # build has no recorder in the process and ignores this. Empty leaves libe3's + # compiled-in default (/tmp/latrec unless the build overrode it). + # + # Rings default to 2^18 records, 8 MiB per thread. The slot path writes five + # records per published slot, so at 2000 slots/s that is ~26 s before the ring + # wraps -- and a wrapped capture is not a valid measurement. Raise it with + # LATREC_ENTRIES_LOG2_E3CONTROLLER_L1_KPM=24 (512 MiB, ~28 min); ceiling 2^28. + latrec_dir: "" +target_slot: -1 diff --git a/include/e3_config.h b/include/e3_config.h new file mode 100644 index 0000000..e2535f0 --- /dev/null +++ b/include/e3_config.h @@ -0,0 +1,227 @@ +/* + * e3_config.h — E3Controller configuration, loaded from a single YAML file. + * + * Replaces the previous 20 command-line options. `--config ` is now the + * only argument, deliberately: two configuration mechanisms invite the two to + * disagree, and this file's whole job is to be the one place the radio geometry + * is stated. + * + * WHY YAML rather than JSON (which would need no new dependency, since libe3 + * already pulls nlohmann): ocudu's own configuration is YAML, including the + * `jbpf:` section this file must agree with. Keeping both sides in the same + * language makes them diffable and reviewable together — which is the point, + * because the controller/gNB geometry agreement is not otherwise enforced by + * anything structural. + */ + +#ifndef E3_CONFIG_H +#define E3_CONFIG_H + +#include +#include +#include + +#include + +namespace e3config { + +/* + * Radio geometry — MUST match the running gNB. + * + * This used to be four `constexpr` values in e3sm_shm_writer.h, with a comment + * requiring lockstep updates with the dApp. Two problems with that: it needed a + * recompile, and it was already inconsistent — `--num-prbs` was accepted on the + * command line and fed only the eCPRI PRB filter, while ShmIqWriter kept sizing + * rows from `constexpr kShmPrbsPerSymbol = 273`. + * + * Treated as a BOOTSTRAP, not as truth: the region has to exist and be sized + * before the codelet can be loaded, so something must be declared up front. But + * the RAN reports its actual geometry in every slot (jbpf_ran_slot_ctx carries + * nof_ports / nof_symbols / nof_subcarriers), so E3SMLayer1 validates these + * values against the first slot it sees and refuses on mismatch. See + * validate_against_ran(). + */ +struct RadioGeometry { + uint16_t nof_ports = 4; /* UL antenna ports */ + uint16_t nof_prbs = 273; /* 100 MHz @ 30 kHz SCS */ + uint16_t nof_symbols = 14; /* per slot */ + uint16_t scs_khz = 30; /* numerology; also fixes slots-per-frame */ + + /* 3GPP: 12 subcarriers per PRB, always. */ + static constexpr uint32_t kScPerPrb = 12; + + /* Complex sample sizes on each side of the conversion. */ + static constexpr uint32_t kBytesPerCbf16 = 4; /* bf16 real + bf16 imag */ + static constexpr uint32_t kBytesPerFp16Pair = 4; /* fp16 real + fp16 imag */ + + uint32_t nof_subcarriers() const { return static_cast(nof_prbs) * kScPerPrb; } + + /* uint16 values per antenna in the fp16 row = 2 (I,Q) per subcarrier. */ + uint32_t u16_per_ant() const { return static_cast(nof_symbols) * nof_subcarriers() * 2u; } + + /* Whole-row uint16 count — this is what SharedMemoryHeader::num_fh_samples + * advertises, and what the dApp multiplies by sizeof(int16_t). */ + uint32_t num_fh_samples() const { return static_cast(nof_ports) * u16_per_ant(); } + + uint32_t row_bytes() const { return num_fh_samples() * 2u; } + + /* Bytes one antenna occupies in the codelet's cbf16 source blob. */ + uint32_t cbf16_bytes_per_ant() const + { + return static_cast(nof_symbols) * nof_subcarriers() * kBytesPerCbf16; + } + + /* Slots per 10 ms radio frame. 20 at 30 kHz, 10 at 15 kHz. This is the range + * of ocudu's slot_point::slot_index(), i.e. of SlotSample::slot_id — NOT + * slots-per-subframe. Getting this wrong is what makes a slot filter select + * nothing. */ + uint32_t slots_per_frame() const { return 10u * (scs_khz / 15u); } + + /* Human-readable, for logs and mismatch diagnostics. */ + std::string describe() const; + + /* Reject nonsense before it becomes a wrong row stride. */ + bool validate(std::string& err) const; +}; + +/* + * TODO: leave only one mode. + * Which process converts the resource grid and writes the fp16 rows. + * Mode: Controller/gnb + * Either way the controller OWNS the region: it creates, sizes, zero-fills and + * headers it, and tears it down. The helper only ever attaches. + */ +enum class ShmWriter { + Controller, + Gnb, +}; + +const char* to_string(ShmWriter w); + +struct ShmConfig { + std::string name = "/e3_ran_buffers"; + size_t size_bytes = static_cast(1) << 30; + + /* Default is Gnb because it is the only mode the SHIPPED CODELET supports: + * uplink_slot_collect.c publishes a ~64 B e3_slot_desc and calls the gNB-side + * helper, so no IQ ever crosses the jbpf ring. Controller mode needs a + * copy-based codelet, which does not currently exist in this repo — selecting + * it yields "Slot blob too small: 0 bytes" on every slot. */ + ShmWriter writer = ShmWriter::Gnb; + + /* + * bf16 -> fp16 scale. Was the E3_CBF16_SCALE environment variable. + * + * Promoted into the config because it is now part of a cross-process + * contract: it is pushed down to the gNB-side publish helper via + * e3_shm_cfg, and if the two writers ever disagreed on it the rows would + * differ with no error at all — bf16 and fp16 are both 2 bytes, so a wrong + * scale is silently wrong data rather than a failure. + */ + float cbf16_scale = 1.0f; +}; + +struct JbpfConfig { + std::string ipc_name = "e3_controller"; + std::string run_path = "/dev/shm"; + size_t mem_size_bytes = static_cast(1) << 30; + std::string lcm_socket_path = "/tmp/jbpf/jbpf_lcm_ipc"; + std::string codelet_base_path = ""; +}; + +struct E3Config { + libe3::EncodingFormat encoding = libe3::EncodingFormat::ASN1; + libe3::E3LinkLayer link_layer = libe3::E3LinkLayer::ZMQ; + libe3::E3TransportLayer transport = libe3::E3TransportLayer::TCP; + uint16_t setup_port = 9990; + uint16_t publisher_port = 9991; + uint16_t subscriber_port = 9999; +}; + +struct ThreadConfig { + int poll_core = -1; + int worker_core = -1; + int publisher_core = -1; + int poll_interval_us = 100; +}; + +struct LoggingConfig { + /* Throttled drop-accounting CSV (<=1 row/s, cumulative). Empty disables it. + * This is the only file the L1-KPM SM writes; per-slot stage timing is + * latrec's job -- see src/e3sm/l1_kpm/l1_kpm_trace.h. */ + std::string drops_log_path; + + /* Per-slot stage CSV for the LEGACY eCPRI Service Model (RF=1) only. + * Empty disables it. + * + * The slot-path SM (RF=2) no longer has one -- see latrec_dir below. This + * one is left as it was: it is on a different data path, and it has the same + * per-slot-flush problem, so it should get the same treatment when that path + * is next touched. */ + std::string spectrum_stats_log_path; + + /* Where latrec writes its per-thread stage-record rings. Empty leaves the + * directory compiled into libe3 (LATREC_DEFAULT_DIR, /tmp/latrec unless the + * build overrode it). + * + * Placement only, NOT a switch: whether anything is recorded is decided + * when libe3 is built, by -DLIBE3_ENABLE_LATREC (./build.sh --latrec). A + * build without it ignores this entirely. */ + std::string latrec_dir; +}; + +struct ControllerConfig { + RadioGeometry radio; + ShmConfig shm; + JbpfConfig jbpf; + E3Config e3; + ThreadConfig threads; + LoggingConfig logging; + + /* + * Legacy single-slot filter, kept for compatibility. + * + * Absolute slot index within a 10 ms frame (0..slots_per_frame()-1). -1 + * disables. NOTE this filters in the controller, AFTER the RAN has already + * converted and published the slot — it saves the controller work and the + * RAN nothing. Workstream F's codelet-side slot_mask supersedes it by making + * the same decision before any data moves; fold this into that when F lands. + */ + int target_slot = -1; +}; + +/* + * Parse a YAML file. Returns false and fills `err` on a parse error, an unknown + * key, or a geometry that fails validate(). Unknown keys are an ERROR rather + * than a warning: a silently ignored `nof_ports` is exactly the class of failure + * this file exists to prevent. + */ +bool load_config(const std::string& path, ControllerConfig& out, std::string& err); + +/* Emit the effective configuration, so what the controller actually used is in + * the log next to the run it produced. */ +void print_config(const ControllerConfig& cfg); + +/* + * Compare the configured geometry against what the RAN reports in a slot. + * + * Returns true when they agree. On disagreement fills `err` with both sides. + * + * The two mismatch directions are NOT symmetric, which is why this exists: + * + * - config UNDER-declares (says 2 ports, gNB sends 4): already safe. The + * gNB-side helper refuses src_len > row_bytes and the codelet reports + * E3_SLOT_FLAG_TRUNCATED, so it is loud. + * + * - config OVER-declares (says 8 ports, gNB sends 4): silently misleading. + * The helper zero-fills the unused antennas, so the dApp reads 4 real ports + * plus 4 of silence and cannot tell without inspecting nof_ports. This is + * the case that produces a plausible-looking wrong spectrum, and the only + * one that needs an explicit check. + */ +bool validate_against_ran(const RadioGeometry& cfg, uint16_t ran_nof_ports, uint16_t ran_nof_symbols, + uint32_t ran_nof_subcarriers, std::string& err); + +} // namespace e3config + +#endif /* E3_CONFIG_H */ diff --git a/include/ecpri_iq_data.h b/include/ecpri_iq_data.h index e1fc0b6..8e9f7db 100644 --- a/include/ecpri_iq_data.h +++ b/include/ecpri_iq_data.h @@ -17,7 +17,16 @@ #define MAX_IQ_PAYLOAD_BYTES 8192 struct iq_sample_data { - uint64_t timestamp; /* Timestamp in nanoseconds */ + /* gNB-side hand-off timestamp (CLOCK_REALTIME ns) stamped by the + * ocudu hook caller right before hook_capture_xran_packet fires. + * RAN anchor for RAN -> codelet latency on this pipeline. Same clock + * domain as codelet_ts_ns below and jbpf_time_get_ns(). */ + uint64_t gnb_ts_ns; + /* Codelet-side timestamp (jbpf_time_get_ns(), CLOCK_REALTIME ns) + * stamped just before jbpf_ringbuf_output. Subtract gnb_ts_ns to + * get gnb_to_codelet_us. */ + uint64_t codelet_ts_ns; + uint64_t timestamp; /* Codelet entry timestamp, us mod 2^31 (kept for ASN.1 IQDataIndication.timestamp) */ uint8_t direction; /* 0=DL, 1=UL */ uint8_t frame_id; /* 3GPP frame ID (0-255, wraps every 2.56s) */ uint8_t comp_method; /* Compression method: 0=none, 1=BFP, 2=block scaling, 3=mu-law, 4=modulation */ @@ -34,7 +43,12 @@ struct iq_sample_data { /* Configuration for PRB-based filtering in the codelet. * Sent via control input channel from the E3Controller. - * Must match the struct in jrtc-apps/codelets/ecpri_iq_samples/ecpri_iq_data.h */ + * + * Must match the struct in codelets/ecpri_iq_samples/ecpri_iq_data.h, which is + * the codelet-side definition and therefore the one that fixes the wire layout + * of the control-input message. Restated here rather than included because this + * header is compiled for the host while that one is compiled for the BPF + * target; if they diverge, the codelet misreads the filter config silently. */ struct prb_filter_config { uint16_t expected_num_prbu; /* 0 = no filtering (pass all), >0 = filter to this PRB count */ }; diff --git a/jbpf_patches/jbpf_monotonic_time.patch b/jbpf_patches/jbpf_monotonic_time.patch new file mode 100644 index 0000000..7d80260 --- /dev/null +++ b/jbpf_patches/jbpf_monotonic_time.patch @@ -0,0 +1,25 @@ +--- a/src/core/jbpf_helper_impl.c ++++ b/src/core/jbpf_helper_impl.c +@@ -328,7 +328,21 @@ + { + struct timespec curr_time = {0, 0}; + uint64_t curr_time_ns = 0; +- clock_gettime(CLOCK_REALTIME, &curr_time); ++ /* CLOCK_MONOTONIC, not CLOCK_REALTIME (LOCAL PATCH -- see ++ * jbpf_monotonic_time.patch in the ocudu_project root). ++ * ++ * This helper stamps the E3 codelet's codelet_ts_ns, which is subtracted ++ * from the gNB's gnb_ts_ns to produce gnb_to_codelet_us. The gNB, the ++ * E3Controller and the dApp all read CLOCK_MONOTONIC now, and it is also the ++ * clock latrec records, so every stage of the path shares one epoch and the ++ * stage CSVs join the latrec rings with no offset arithmetic. ++ * ++ * Monotonic additionally cannot be stepped by NTP, which REALTIME can -- ++ * a step mid-run silently corrupts every stage derived from these stamps. ++ * ++ * Caveat: monotonic shares an origin only within one boot, so stamps are ++ * comparable across processes on ONE host, not across hosts. */ ++ clock_gettime(CLOCK_MONOTONIC, &curr_time); + curr_time_ns = curr_time.tv_nsec + curr_time.tv_sec * 1.0e9; + return curr_time_ns; + } diff --git a/libe3 b/libe3 index 1451e9a..002c843 160000 --- a/libe3 +++ b/libe3 @@ -1 +1 @@ -Subproject commit 1451e9a008a3fd24ee75d6e0c81bb01045235347 +Subproject commit 002c843a107da1be71ea22c60d7c3cbca14ba296 diff --git a/src/e3_config.cpp b/src/e3_config.cpp new file mode 100644 index 0000000..076472a --- /dev/null +++ b/src/e3_config.cpp @@ -0,0 +1,419 @@ +/* + * e3_config.cpp — YAML loader for the E3Controller configuration. + */ + +#include "e3_config.h" + +#include + +#include +#include +#include + +#include + +namespace e3config { +namespace { + +/* + * Unknown keys are an ERROR, not a warning. + * + * A typo'd `nof_port:` that is silently ignored leaves the controller sizing + * rows for the default 4 ports while the operator believes they configured + * something else — which is precisely the silent-geometry-mismatch failure this + * config exists to eliminate. Better to refuse to start. + */ +bool +check_keys(const YAML::Node& node, const char* section, const std::set& allowed, std::string& err) +{ + if (!node || !node.IsMap()) { + return true; + } + for (const auto& kv : node) { + const std::string k = kv.first.as(); + if (allowed.find(k) == allowed.end()) { + std::ostringstream os; + os << "unknown key '" << k << "' in section '" << section << "'"; + err = os.str(); + return false; + } + } + return true; +} + +template +void +get(const YAML::Node& n, const char* key, T& out) +{ + if (n && n[key]) { + out = n[key].as(); + } +} + +bool +parse_encoding(const std::string& s, libe3::EncodingFormat& out, std::string& err) +{ + if (s == "asn1" || s == "ASN1" || s == "aper") { + out = libe3::EncodingFormat::ASN1; + return true; + } + if (s == "json" || s == "JSON") { + out = libe3::EncodingFormat::JSON; + return true; + } + err = "e3.encoding must be 'asn1' or 'json', got '" + s + "'"; + return false; +} + +bool +parse_link_layer(const std::string& s, libe3::E3LinkLayer& out, std::string& err) +{ + if (s == "zmq" || s == "ZMQ") { + out = libe3::E3LinkLayer::ZMQ; + return true; + } + if (s == "posix" || s == "POSIX") { + out = libe3::E3LinkLayer::POSIX; + return true; + } + err = "e3.link_layer must be 'zmq' or 'posix', got '" + s + "'"; + return false; +} + +bool +parse_transport(const std::string& s, libe3::E3TransportLayer& out, std::string& err) +{ + if (s == "tcp" || s == "TCP") { + out = libe3::E3TransportLayer::TCP; + return true; + } + if (s == "ipc" || s == "IPC") { + out = libe3::E3TransportLayer::IPC; + return true; + } + if (s == "sctp" || s == "SCTP") { + out = libe3::E3TransportLayer::SCTP; + return true; + } + err = "e3.transport must be 'tcp', 'ipc' or 'sctp', got '" + s + "'"; + return false; +} + +bool +parse_writer(const std::string& s, ShmWriter& out, std::string& err) +{ + if (s == "controller") { + out = ShmWriter::Controller; + return true; + } + if (s == "gnb") { + out = ShmWriter::Gnb; + return true; + } + err = "shm.writer must be 'controller' or 'gnb', got '" + s + "'"; + return false; +} + +} // namespace + +const char* +to_string(ShmWriter w) +{ + return w == ShmWriter::Gnb ? "gnb" : "controller"; +} + +std::string +RadioGeometry::describe() const +{ + std::ostringstream os; + os << nof_ports << " ports x " << nof_symbols << " sym x " << nof_subcarriers() << " subc (" << nof_prbs + << " PRB, " << scs_khz << " kHz SCS)"; + return os.str(); +} + +bool +RadioGeometry::validate(std::string& err) const +{ + if (nof_ports == 0 || nof_ports > 8) { + err = "radio.nof_ports must be 1..8"; + return false; + } + if (nof_prbs == 0 || nof_prbs > 275) { + err = "radio.nof_prbs must be 1..275"; + return false; + } + if (nof_symbols != 12 && nof_symbols != 14) { + err = "radio.nof_symbols must be 12 or 14"; + return false; + } + if (scs_khz != 15 && scs_khz != 30 && scs_khz != 60 && scs_khz != 120) { + err = "radio.scs_khz must be 15, 30, 60 or 120"; + return false; + } + /* + * The gNB-side convert loop processes 8 uint16 lanes per AVX2 iteration and + * relies on there being no tail. 273 PRB x 14 sym x 2 gives 91,728 = 8 x + * 11,466. A geometry that breaks that would silently take the scalar tail + * path for the remainder, so refuse it here rather than discover it in the + * PHY RX thread. + */ + if ((u16_per_ant() % 8u) != 0u) { + err = "radio geometry gives " + std::to_string(u16_per_ant()) + + " uint16/antenna, which is not a multiple of 8 (required by the AVX2 convert)"; + return false; + } + return true; +} + +bool +load_config(const std::string& path, ControllerConfig& out, std::string& err) +{ + /* + * Start from defaults rather than from whatever the caller passed in. + * + * Keys absent from the YAML are left untouched by design (that is what makes + * them optional), so without this reset a reused ControllerConfig would + * inherit fields from a previous load — and on failure the caller would hold + * a half-populated config that looks usable. Partially-applied configuration + * is precisely the kind of silent wrongness this file exists to prevent. + */ + out = ControllerConfig{}; + + YAML::Node root; + try { + root = YAML::LoadFile(path); + } catch (const std::exception& e) { + err = std::string("cannot load '") + path + "': " + e.what(); + return false; + } + + if (!root.IsMap()) { + err = "top level of '" + path + "' must be a mapping"; + return false; + } + + if (!check_keys(root, "", {"radio", "shm", "jbpf", "e3", "threads", "logging", "target_slot"}, err)) { + return false; + } + + bool lcm_explicit = false; + + try { + /* ---- radio ---- */ + if (const auto n = root["radio"]) { + if (!check_keys(n, "radio", {"nof_ports", "nof_prbs", "nof_symbols", "scs_khz"}, err)) { + return false; + } + get(n, "nof_ports", out.radio.nof_ports); + get(n, "nof_prbs", out.radio.nof_prbs); + get(n, "nof_symbols", out.radio.nof_symbols); + get(n, "scs_khz", out.radio.scs_khz); + } + + /* ---- shm ---- */ + if (const auto n = root["shm"]) { + if (!check_keys(n, "shm", {"name", "size_bytes", "cbf16_scale", "writer"}, err)) { + return false; + } + get(n, "name", out.shm.name); + get(n, "size_bytes", out.shm.size_bytes); + get(n, "cbf16_scale", out.shm.cbf16_scale); + if (n["writer"] && !parse_writer(n["writer"].as(), out.shm.writer, err)) { + return false; + } + } + + /* ---- jbpf ---- */ + if (const auto n = root["jbpf"]) { + if (!check_keys( + n, "jbpf", + {"ipc_name", "run_path", "mem_size_bytes", "lcm_socket_path", "codelet_base_path"}, err)) { + return false; + } + get(n, "ipc_name", out.jbpf.ipc_name); + get(n, "run_path", out.jbpf.run_path); + get(n, "mem_size_bytes", out.jbpf.mem_size_bytes); + /* Derive from run_path unless explicitly set. The gNB composes + * //, so an + * independent absolute default here (it used to be /tmp/...) drifts + * from the gNB the moment run_path is anything but /tmp — which is + * exactly what happens in the standard /dev/shm deployment. */ + if (n["lcm_socket_path"]) { + out.jbpf.lcm_socket_path = n["lcm_socket_path"].as(); + lcm_explicit = true; + } + get(n, "codelet_base_path", out.jbpf.codelet_base_path); + } + + /* ---- e3 ---- */ + if (const auto n = root["e3"]) { + if (!check_keys(n, "e3", + {"encoding", "link_layer", "transport", "setup_port", "publisher_port", + "subscriber_port"}, + err)) { + return false; + } + if (n["encoding"] && !parse_encoding(n["encoding"].as(), out.e3.encoding, err)) { + return false; + } + if (n["link_layer"] && !parse_link_layer(n["link_layer"].as(), out.e3.link_layer, err)) { + return false; + } + if (n["transport"] && !parse_transport(n["transport"].as(), out.e3.transport, err)) { + return false; + } + get(n, "setup_port", out.e3.setup_port); + get(n, "publisher_port", out.e3.publisher_port); + get(n, "subscriber_port", out.e3.subscriber_port); + } + + /* ---- threads ---- */ + if (const auto n = root["threads"]) { + if (!check_keys(n, "threads", {"poll_core", "worker_core", "publisher_core", "poll_interval_us"}, + err)) { + return false; + } + get(n, "poll_core", out.threads.poll_core); + get(n, "worker_core", out.threads.worker_core); + get(n, "publisher_core", out.threads.publisher_core); + get(n, "poll_interval_us", out.threads.poll_interval_us); + } + + /* ---- logging ---- */ + if (const auto n = root["logging"]) { + if (!check_keys(n, "logging", + {"drops_log_path", "spectrum_stats_log_path", "latrec_dir"}, err)) { + return false; + } + get(n, "drops_log_path", out.logging.drops_log_path); + get(n, "spectrum_stats_log_path", out.logging.spectrum_stats_log_path); + get(n, "latrec_dir", out.logging.latrec_dir); + } + + get(root, "target_slot", out.target_slot); + } catch (const std::exception& e) { + err = std::string("error parsing '") + path + "': " + e.what(); + return false; + } + + if (!lcm_explicit) { + out.jbpf.lcm_socket_path = out.jbpf.run_path + "/jbpf/jbpf_lcm_ipc"; + } + + if (!out.radio.validate(err)) { + return false; + } + + /* The SHM region must hold at least one row per buffer, or the ring degenerates. */ + const size_t kHeaderBytes = 64; + const size_t kNumBuffers = 2; + if (out.shm.size_bytes <= kHeaderBytes || + (out.shm.size_bytes - kHeaderBytes) / kNumBuffers < out.radio.row_bytes()) { + std::ostringstream os; + os << "shm.size_bytes " << out.shm.size_bytes << " is too small for " << kNumBuffers << " buffers of one " + << out.radio.row_bytes() << "-byte row (" << out.radio.describe() << ")"; + err = os.str(); + return false; + } + + if (!(out.shm.cbf16_scale > 0.0f)) { + err = "shm.cbf16_scale must be > 0 (a zero or negative scale zeroes the row and " + "silently breaks the dApp)"; + return false; + } + + if (out.target_slot >= 0 && static_cast(out.target_slot) >= out.radio.slots_per_frame()) { + std::ostringstream os; + os << "target_slot " << out.target_slot << " is outside 0.." << (out.radio.slots_per_frame() - 1) + << " for " << out.radio.scs_khz << " kHz SCS — it would never match"; + err = os.str(); + return false; + } + + return true; +} + +void +print_config(const ControllerConfig& cfg) +{ + std::printf("E3Controller configuration:\n"); + std::printf(" radio: %s\n", cfg.radio.describe().c_str()); + std::printf(" row = %u uint16 (%u bytes), %u slots/frame\n", cfg.radio.num_fh_samples(), + cfg.radio.row_bytes(), cfg.radio.slots_per_frame()); + std::printf(" shm: %s, %zu bytes, cbf16_scale=%g\n", cfg.shm.name.c_str(), cfg.shm.size_bytes, + static_cast(cfg.shm.cbf16_scale)); + std::printf(" row writer: %s%s\n", to_string(cfg.shm.writer), + cfg.shm.writer == ShmWriter::Gnb + ? " (gNB helper converts + writes; controller owns the region only)" + : " (controller converts + writes)"); + std::printf(" jbpf: ipc=%s run=%s lcm=%s\n", cfg.jbpf.ipc_name.c_str(), cfg.jbpf.run_path.c_str(), + cfg.jbpf.lcm_socket_path.c_str()); + std::printf(" codelets: %s\n", + cfg.jbpf.codelet_base_path.empty() ? "(none — no auto-loading)" + : cfg.jbpf.codelet_base_path.c_str()); + std::printf(" e3: ports %u/%u/%u\n", cfg.e3.setup_port, cfg.e3.publisher_port, + cfg.e3.subscriber_port); + std::printf(" threads: poll=%d worker=%d publisher=%d\n", cfg.threads.poll_core, cfg.threads.worker_core, + cfg.threads.publisher_core); + if (cfg.target_slot >= 0) { + std::printf(" target_slot: %d (controller-side filter)\n", cfg.target_slot); + } + if (!cfg.logging.drops_log_path.empty()) { + std::printf(" drop log: %s\n", cfg.logging.drops_log_path.c_str()); + } + /* Three distinguishable states, worth telling apart on startup: not + * compiled in, compiled in and writing where libe3 was built to write, or + * compiled in and redirected. Otherwise "I set latrec_dir and got no rings" + * is indistinguishable from a wrong path. */ +#ifdef LIBE3_ENABLE_LATREC + std::printf(" stage recs: enabled -> %s\n", + cfg.logging.latrec_dir.empty() + ? LATREC_DEFAULT_DIR " (libe3 default)" + : cfg.logging.latrec_dir.c_str()); +#else + if (!cfg.logging.latrec_dir.empty()) { + std::printf(" stage recs: IGNORED (%s): libe3 built without " + "-DLIBE3_ENABLE_LATREC\n", cfg.logging.latrec_dir.c_str()); + } else { + std::printf(" stage recs: not compiled in\n"); + } +#endif +} + +bool +validate_against_ran(const RadioGeometry& cfg, uint16_t ran_nof_ports, uint16_t ran_nof_symbols, + uint32_t ran_nof_subcarriers, std::string& err) +{ + std::ostringstream os; + + /* The dangerous direction: we advertise more antennas than the RAN delivers, + * so the surplus is written as silence and the dApp cannot distinguish it + * from a genuinely quiet antenna. */ + if (ran_nof_ports != 0 && cfg.nof_ports > ran_nof_ports) { + os << "config declares " << cfg.nof_ports << " antenna ports but the RAN reports " << ran_nof_ports + << ". The extra " << (cfg.nof_ports - ran_nof_ports) + << " antenna(s) would be published as silence and the dApp could not tell. " + "Set radio.nof_ports to " << ran_nof_ports << "."; + err = os.str(); + return false; + } + + if (ran_nof_symbols != 0 && cfg.nof_symbols != ran_nof_symbols) { + os << "config declares " << cfg.nof_symbols << " symbols/slot but the RAN reports " << ran_nof_symbols + << " — row stride would be wrong."; + err = os.str(); + return false; + } + + if (ran_nof_subcarriers != 0 && cfg.nof_subcarriers() != ran_nof_subcarriers) { + os << "config declares " << cfg.nof_subcarriers() << " subcarriers (" << cfg.nof_prbs + << " PRB) but the RAN reports " << ran_nof_subcarriers << " (" << (ran_nof_subcarriers / 12) + << " PRB) — row stride would be wrong."; + err = os.str(); + return false; + } + + return true; +} + +} // namespace e3config diff --git a/src/e3_controller.cpp b/src/e3_controller.cpp index 4a18109..39ed6d9 100644 --- a/src/e3_controller.cpp +++ b/src/e3_controller.cpp @@ -22,14 +22,17 @@ #include #include #include +#include #include #include #include #include +#include "libe3_stamp.h" // generated: which libe3 we linked against #if defined(__x86_64__) || defined(__i386__) #include #endif #include +#include extern "C" { #include "jbpf_io.h" @@ -38,6 +41,7 @@ extern "C" { #include "jbpf_mem_mgmt.h" } +#include "e3_config.h" #include "jbpf_dispatcher.h" #include "e3sm/sm_spectrum/e3sm_spectrum.h" #include "e3sm/l1_kpm/e3sm_layer_1.h" @@ -137,311 +141,192 @@ static void signal_handler(int signo) // } // ---- Configuration ---- - -struct E3ControllerConfig { - std::string ipc_name = "e3_controller"; - std::string run_path = "/dev/shm"; - size_t mem_size = JBPF_HUGEPAGE_SIZE_1GB; - int poll_interval_us = 100; // microseconds between polls (ignored when poll_core >= 0) - int poll_core = -1; // CPU to pin the polling thread to (-1 = no pinning, sleep-based) - int worker_core = -1; // CPU to pin the SM worker thread to (-1 = no pinning) - int publisher_core = -1; // CPU to pin libe3's publisher_thread_ to (-1 = no pinning) - uint16_t num_prbs = 106; // expected number of PRBs per symbol - std::string lcm_socket_path = "/tmp/jbpf/jbpf_lcm_ipc"; // LCM IPC socket for codelet loading - std::string codelet_base_path; // base dir for codelet binaries (empty = no auto-loading) - - // Wire encoding for the (single) E3 channel. The controller serves exactly - // one encoding at a time; dApps must speak the same one. ASN.1 (APER) is - // the O-RAN default; JSON matches the NVIDIA_L1 cuBB convention used by the - // adaptive_cpu dApp. - // - // libe3 is built with BOTH encoders (LIBE3_ENABLE_ASN1 + LIBE3_ENABLE_JSON), - // so this is a pure runtime choice — the encoder factory picks the matching - // encoder. (If libe3 was instead built with only one encoder, this must - // match it or the outbound encoder rejects the PDU.) - libe3::EncodingFormat encoding = libe3::EncodingFormat::ASN1; - - // Link layer libe3 uses to move E3AP PDUs. ZMQ is the default (and the only - // one exercised here); POSIX selects raw sockets. - libe3::E3LinkLayer link_layer = libe3::E3LinkLayer::ZMQ; - - // Transport under the link layer. TCP matches what most dApps expect - // (host+ports); IPC uses UNIX-domain sockets for a same-host dApp; SCTP is - // the O-RAN standard. - libe3::E3TransportLayer transport = libe3::E3TransportLayer::TCP; - - // E3 channel ports (libe3 defaults; for a JSON/cuBB dApp use 5555/5556/5557). - uint16_t setup_port = 9990; - uint16_t publisher_port = 9991; - uint16_t subscriber_port = 9999; - - // POSIX SHM segment for IQ data. Layout matches SharedMemoryHeader in - // spear-aerial-sample-apps' e3_manager.h so the dApp reads our data with no - // code change. - std::string shm_name = "/e3_ran_buffers"; - size_t shm_size = static_cast(1) << 30; // 1 GiB - - // --- Timing logs (all disabled unless a path is given) --- - // Per-slot RAN-side stage CSV written by E3SMLayer1 (gnb→codelet→dispatch→ - // handler durations + shm/encode/emit costs). Empty = no log. - std::string stats_log_path; - - // Optional single-slot filter. The value is the absolute slot index within - // a 10 ms frame (0..19 for 30 kHz SCS, 0..9 for 15 kHz). UL symbols whose - // computed (subframe_id*2 + slot_id) doesn't match are dropped at the top - // of process_sample_json — no BFP decompress, no slot accumulation. -1 - // (default) disables the filter and every UL slot is forwarded. - // NOTE: assumes 30 kHz SCS (2 slots/subframe). For a different numerology - // the divider would need to change. - int target_slot = -1; -}; +// +// The controller is configured by a single YAML file. This replaced 20 +// command-line options; `--config` is deliberately the only argument, because +// two configuration mechanisms invite the two to disagree, and the radio +// geometry in that file has to be the one place it is stated. +// +// See include/e3_config.h for the schema and for why the geometry is treated as +// a bootstrap that gets validated against the RAN rather than as truth. static void print_usage(const char* prog) { - std::printf("Usage: %s [options]\n" - "Options:\n" - " --ipc-name IPC shared memory name (default: e3_controller)\n" - " --run-path jbpf run path (default: /dev/shm)\n" - " --mem-size Shared memory size in bytes (default: 1GB)\n" - " --poll-interval Poll interval in microseconds (default: 100; ignored if --poll-core is set)\n" - " --poll-core Pin the polling thread to and busy-poll (default: -1, no pinning)\n" - " --worker-core Pin the SM worker (decompress/encode/emit) to (default: -1, no pinning)\n" - " --publisher-core Pin libe3's publisher thread (outer encode + ZMQ send) to (default: -1, no pinning)\n" - " --num-prbs Expected number of PRBs per symbol (default: 106)\n" - " --lcm-socket LCM IPC socket path for codelet loading (default: /tmp/jbpf/jbpf_lcm_ipc)\n" - " --codelet-path Base directory for codelet binaries (enables auto-loading)\n" - " --encoding Wire encoding for the E3 channel: 'asn1' (default) or\n" - " 'json'. Must match how libe3 was compiled.\n" - " --link-layer Link layer: 'zmq' (default) or 'posix'.\n" - " --transport Transport layer: 'tcp' (default), 'ipc', or 'sctp'.\n" - " --setup-port

E3 channel setup REP port (default: 9990)\n" - " --publisher-port

E3 channel indication PUB port (default: 9991)\n" - " --subscriber-port

E3 channel control SUB port (default: 9999)\n" - " --shm-name POSIX SHM name for IQ data (default: /e3_ran_buffers)\n" - " --shm-size POSIX SHM size in bytes (default: 1GiB)\n" - " --target-slot Forward ONLY UL slot N (absolute slot within a frame,\n" - " 0..19 for 30 kHz SCS). Assumes 2 slots/subframe.\n" - " Default -1 = forward every UL slot.\n" - " --stats-log Write the per-slot RAN-side stage CSV to \n" - " (gnb/codelet/dispatch/handler + shm/encode/emit).\n" - " Default: disabled.\n" - " --help Show this help\n", + std::printf("Usage: %s --config \n" + "\n" + " --config YAML configuration (required)\n" + " --help this message\n" + "\n" + "See configs/e3_controller.yaml for a documented example.\n", prog); } -static E3ControllerConfig parse_args(int argc, char** argv) +/* Report which libe3 this binary was linked against, and shout if /usr/local has + * moved on since. + * + * libe3 is linked STATICALLY (libe3::libe3 is STATIC IMPORTED -> liblibe3.a), so + * `cmake --install libe3/build` has no effect on an already-built controller and + * leaves no runtime trace. Editing SLEEP_DURATION, reinstalling, and forgetting + * to relink yields a controller that still has the old value with nothing to + * indicate it -- the only symptom is a queue-wait distribution that quietly + * refuses to move. Comparing the .a's mtime against our own is enough to catch + * it, and needs no build-time bookkeeping beyond the stamp header. */ +static void report_libe3_provenance() { - E3ControllerConfig config; - - // Long-only options (no short flag) get codes >= 256 so they don't collide - // with single-char optstring values. - enum LongOpt { - OPT_TRANSPORT = 256, - OPT_ENCODING, - OPT_LINK_LAYER, - OPT_SETUP_PORT, - OPT_PUBLISHER_PORT, - OPT_SUBSCRIBER_PORT, - OPT_SHM_NAME, - OPT_SHM_SIZE, - OPT_TARGET_SLOT, - OPT_STATS_LOG, - }; + std::cout << "[libe3] linked " << LIBE3_STAMP_VERSION + << " static, SLEEP_DURATION=" << LIBE3_STAMP_SLEEP_US << "us" + << ", build type '" << LIBE3_STAMP_BUILDTYPE << "'\n"; + if (std::strlen(LIBE3_STAMP_BUILDTYPE) == 0) { + std::cout << "[libe3] WARNING: libe3 was built with an EMPTY CMAKE_BUILD_TYPE," + " i.e. no -O flag.\n" + " The E3AP encoder, connector and outbound queue are" + " unoptimised; latency\n" + " numbers from this build are not comparable with a" + " Release one. Rebuild:\n" + " cmake -S libe3 -B libe3/build -DCMAKE_BUILD_TYPE=Release" + " && cmake --build libe3/build -j\n" + " cmake --install libe3/build && cmake --build build" + " --target e3_controller\n"; + } - static struct option long_options[] = { - {"ipc-name", required_argument, nullptr, 'n'}, - {"run-path", required_argument, nullptr, 'r'}, - {"mem-size", required_argument, nullptr, 'm'}, - {"poll-interval", required_argument, nullptr, 'p'}, - {"poll-core", required_argument, nullptr, 'C'}, - {"worker-core", required_argument, nullptr, 'W'}, - {"publisher-core", required_argument, nullptr, 'P'}, - {"num-prbs", required_argument, nullptr, 'b'}, - {"lcm-socket", required_argument, nullptr, 'l'}, - {"codelet-path", required_argument, nullptr, 'c'}, - {"encoding", required_argument, nullptr, OPT_ENCODING}, - {"link-layer", required_argument, nullptr, OPT_LINK_LAYER}, - {"transport", required_argument, nullptr, OPT_TRANSPORT}, - {"setup-port", required_argument, nullptr, OPT_SETUP_PORT}, - {"publisher-port", required_argument, nullptr, OPT_PUBLISHER_PORT}, - {"subscriber-port", required_argument, nullptr, OPT_SUBSCRIBER_PORT}, - {"shm-name", required_argument, nullptr, OPT_SHM_NAME}, - {"shm-size", required_argument, nullptr, OPT_SHM_SIZE}, - {"target-slot", required_argument, nullptr, OPT_TARGET_SLOT}, - {"stats-log", required_argument, nullptr, OPT_STATS_LOG}, - {"help", no_argument, nullptr, 'h'}, - {nullptr, 0, nullptr, 0 } - }; + struct stat lib{}, self{}; + if (::stat(LIBE3_STAMP_LIB, &lib) != 0) { return; } // not installed there; nothing to compare + if (::stat("/proc/self/exe", &self) != 0) { return; } + if (lib.st_mtime > self.st_mtime) { + std::cout << "[libe3] WARNING: " << LIBE3_STAMP_LIB << " is NEWER than this" + " binary.\n" + " libe3 was reinstalled after this controller was linked," + " so you are running\n" + " a STALE snapshot -- the reinstalled changes are NOT in" + " this process. Relink:\n" + " cmake --build build --target e3_controller\n"; + } +} - int opt; - while ((opt = getopt_long(argc, argv, "n:r:m:p:C:W:P:b:l:c:h", long_options, nullptr)) != -1) { - switch (opt) { - case 'n': - config.ipc_name = optarg; - break; - case 'r': - config.run_path = optarg; - break; - case 'm': - config.mem_size = std::strtoull(optarg, nullptr, 0); - break; - case 'p': - config.poll_interval_us = std::atoi(optarg); - break; - case 'C': - config.poll_core = std::atoi(optarg); - break; - case 'W': - config.worker_core = std::atoi(optarg); - break; - case 'P': - config.publisher_core = std::atoi(optarg); - break; - case 'b': - config.num_prbs = static_cast(std::atoi(optarg)); - break; - case 'l': - config.lcm_socket_path = optarg; - break; - case 'c': - config.codelet_base_path = optarg; - break; - case OPT_ENCODING: { - std::string v(optarg); - if (v == "asn1" || v == "ASN1" || v == "aper" || v == "APER") { - config.encoding = libe3::EncodingFormat::ASN1; - } else if (v == "json" || v == "JSON") { - config.encoding = libe3::EncodingFormat::JSON; - } else { - std::fprintf(stderr, "[E3Controller] Unknown --encoding value '%s' (expected asn1 or json)\n", - optarg); - std::exit(1); - } - break; - } - case OPT_LINK_LAYER: { - std::string v(optarg); - if (v == "zmq" || v == "ZMQ") { - config.link_layer = libe3::E3LinkLayer::ZMQ; - } else if (v == "posix" || v == "POSIX") { - config.link_layer = libe3::E3LinkLayer::POSIX; - } else { - std::fprintf(stderr, "[E3Controller] Unknown --link-layer value '%s' (expected zmq or posix)\n", - optarg); - std::exit(1); - } - break; - } - case OPT_TRANSPORT: { - std::string v(optarg); - if (v == "tcp" || v == "TCP") { - config.transport = libe3::E3TransportLayer::TCP; - } else if (v == "ipc" || v == "IPC") { - config.transport = libe3::E3TransportLayer::IPC; - } else if (v == "sctp" || v == "SCTP") { - config.transport = libe3::E3TransportLayer::SCTP; - } else { - std::fprintf(stderr, "[E3Controller] Unknown --transport value '%s' (expected tcp, ipc, or sctp)\n", - optarg); - std::exit(1); - } - break; - } - case OPT_SETUP_PORT: - config.setup_port = static_cast(std::atoi(optarg)); - break; - case OPT_PUBLISHER_PORT: - config.publisher_port = static_cast(std::atoi(optarg)); - break; - case OPT_SUBSCRIBER_PORT: - config.subscriber_port = static_cast(std::atoi(optarg)); - break; - case OPT_SHM_NAME: - config.shm_name = optarg; - break; - case OPT_SHM_SIZE: - config.shm_size = std::strtoull(optarg, nullptr, 0); - break; - case OPT_TARGET_SLOT: - config.target_slot = std::atoi(optarg); - break; - case OPT_STATS_LOG: - config.stats_log_path = optarg; - break; - case 'h': - default: +int main(int argc, char** argv) +{ + report_libe3_provenance(); + + // Single argument by design; see print_usage. + std::string config_path; + for (int i = 1; i < argc; ++i) { + if (std::strcmp(argv[i], "--config") == 0 && i + 1 < argc) { + config_path = argv[++i]; + } else if (std::strcmp(argv[i], "--help") == 0 || std::strcmp(argv[i], "-h") == 0) { + print_usage(argv[0]); + return 0; + } else { + std::fprintf(stderr, "unexpected argument '%s'\n\n", argv[i]); print_usage(argv[0]); - std::exit(opt == 'h' ? 0 : 1); + return 1; } } + if (config_path.empty()) { + std::fprintf(stderr, "--config is required\n\n"); + print_usage(argv[0]); + return 1; + } - return config; -} - -// ---- Main ---- + e3config::ControllerConfig config; + std::string cfg_err; + if (!e3config::load_config(config_path, config, cfg_err)) { + std::fprintf(stderr, "[E3Controller] configuration error: %s\n", cfg_err.c_str()); + return 1; + } + /* Where the stage-record rings go. This has to run before any thread opens + * one: the directory is read at each ring open, and both libe3's agent + * threads and our own pipeline worker open theirs as they start up. A no-op + * in a build without the recorder, so it is unconditional. */ + if (!config.logging.latrec_dir.empty()) { + latrec_set_output_dir(config.logging.latrec_dir.c_str()); + } -int main(int argc, char** argv) -{ - E3ControllerConfig config = parse_args(argc, argv); + e3config::print_config(config); + + /* The LCM socket is created by the gNB, so its absence usually means the gNB + * is not up yet or the two disagree on the path. Say which, here, rather than + * letting it surface later as a codeletset load failure whose message + * ("is srsRAN running?") points at the wrong thing. + * + * Not fatal: the controller is the IPC PRIMARY and is supposed to start + * first, so at this point the gNB legitimately may not exist yet. */ + { + struct stat st; + if (::stat(config.jbpf.lcm_socket_path.c_str(), &st) != 0) { + std::fprintf(stderr, + "[E3Controller] NOTE: LCM socket '%s' does not exist yet.\n" + " Codelet loading will fail until it appears. The gNB composes this path as\n" + " // from its own YAML — with\n" + " the usual /dev/shm + jbpf + jbpf_lcm_ipc that is /dev/shm/jbpf/jbpf_lcm_ipc.\n" + " If the gNB is already running, jbpf.run_path here (%s) disagrees with it.\n", + config.jbpf.lcm_socket_path.c_str(), config.jbpf.run_path.c_str()); + } + } // E3 agent configuration — single encoding on a single channel. std::string ran_id = "ocudu-janus"; libe3::E3Config agentConfig; agentConfig.ran_identifier = ran_id; - agentConfig.link_layer = config.link_layer; - agentConfig.transport_layer = config.transport; + agentConfig.link_layer = config.e3.link_layer; + agentConfig.transport_layer = config.e3.transport; // One encoding, chosen via --encoding. dApps must speak the same encoding // and connect to setup/publisher/subscriber ports below. - agentConfig.encoding = config.encoding; - agentConfig.setup_port = config.setup_port; - agentConfig.publisher_port = config.publisher_port; - agentConfig.subscriber_port = config.subscriber_port; + agentConfig.encoding = config.e3.encoding; + agentConfig.setup_port = config.e3.setup_port; + agentConfig.publisher_port = config.e3.publisher_port; + agentConfig.subscriber_port = config.e3.subscriber_port; agentConfig.log_level = 5; // 0=none, 1=err, 2=warn, 3=info, 4=debug, 5=trace // Pin libe3's I/O threads (the RAN outbound loop in particular — it does // the outer encode + zmq_send) to a dedicated core if requested. Without - // this the publisher gets scheduled out under load and queue_us in the - // publisher-stage CSV climbs into hundreds of µs. - if (config.publisher_core >= 0) { - agentConfig.io_thread_affinity = config.publisher_core; + // this the publisher gets scheduled out under load and it can be + // scheduled out for hundreds of µs. + if (config.threads.publisher_core >= 0) { + agentConfig.io_thread_affinity = config.threads.publisher_core; } auto encoding_name = [](libe3::EncodingFormat e) { return e == libe3::EncodingFormat::JSON ? "JSON" : "ASN.1 (APER)"; }; + // Per-slot stage CSV for the legacy eCPRI SM (RF=1) only. Its own config + // key now rather than a suffix derived from the slot-path SM's, because the + // slot-path SM no longer writes one -- see logging.latrec_dir. + const std::string& spectrum_stats_log_path = config.logging.spectrum_stats_log_path; + std::cout << "=============================================\n"; std::cout << "E3 Agent Configuration (single-encoding):\n" << " RAN ID: " << ran_id << "\n" - << " Encoding: " << encoding_name(config.encoding) << "\n" - << " Link layer: " << libe3::link_layer_to_string(config.link_layer) << "\n" - << " Transport: " << libe3::transport_layer_to_string(config.transport) << "\n" - << " Channel: setup=" << config.setup_port - << " publisher=" << config.publisher_port - << " subscriber=" << config.subscriber_port << "\n" - << " Stats log: " - << (config.stats_log_path.empty() ? "(disabled)" : config.stats_log_path) << "\n\n"; + << " Encoding: " << encoding_name(config.e3.encoding) << "\n" + << " Link layer: " << libe3::link_layer_to_string(config.e3.link_layer) << "\n" + << " Transport: " << libe3::transport_layer_to_string(config.e3.transport) << "\n" + << " Channel: setup=" << config.e3.setup_port + << " publisher=" << config.e3.publisher_port + << " subscriber=" << config.e3.subscriber_port << "\n" + << " Spectrum stats: " + << (spectrum_stats_log_path.empty() ? "(disabled)" : spectrum_stats_log_path) + << "\n\n"; libe3::E3Agent agent(std::move(agentConfig)); std::printf("=============================================\n"); std::printf(" E3Controller Codelet Configuration\n"); std::printf("=============================================\n"); - std::printf(" IPC name: %s\n", config.ipc_name.c_str()); - std::printf(" Run path: %s\n", config.run_path.c_str()); - std::printf(" Memory size: %zu bytes\n", config.mem_size); - std::printf(" Poll interval: %d us%s\n", config.poll_interval_us, - config.poll_core >= 0 ? " (ignored — busy poll on dedicated core)" : ""); + std::printf(" IPC name: %s\n", config.jbpf.ipc_name.c_str()); + std::printf(" Run path: %s\n", config.jbpf.run_path.c_str()); + std::printf(" Memory size: %zu bytes\n", config.jbpf.mem_size_bytes); + std::printf(" Poll interval: %d us%s\n", config.threads.poll_interval_us, + config.threads.poll_core >= 0 ? " (ignored — busy poll on dedicated core)" : ""); std::printf(" Poll core: %s\n", - config.poll_core >= 0 ? std::to_string(config.poll_core).c_str() : "(none — sleep-based)"); + config.threads.poll_core >= 0 ? std::to_string(config.threads.poll_core).c_str() : "(none — sleep-based)"); std::printf(" Worker core: %s\n", - config.worker_core >= 0 ? std::to_string(config.worker_core).c_str() : "(none)"); + config.threads.worker_core >= 0 ? std::to_string(config.threads.worker_core).c_str() : "(none)"); std::printf(" Publisher core: %s\n", - config.publisher_core >= 0 ? std::to_string(config.publisher_core).c_str() : "(none)"); - std::printf(" Num PRBs: %u\n", config.num_prbs); - std::printf(" LCM socket: %s\n", config.lcm_socket_path.c_str()); + config.threads.publisher_core >= 0 ? std::to_string(config.threads.publisher_core).c_str() : "(none)"); + std::printf(" Num PRBs: %u\n", config.radio.nof_prbs); + std::printf(" LCM socket: %s\n", config.jbpf.lcm_socket_path.c_str()); std::printf(" Codelet path: %s\n", - config.codelet_base_path.empty() ? "(none — auto-loading disabled)" : config.codelet_base_path.c_str()); + config.jbpf.codelet_base_path.empty() ? "(none — auto-loading disabled)" : config.jbpf.codelet_base_path.c_str()); std::printf("=============================================\n\n"); // Install signal handlers @@ -455,18 +340,18 @@ int main(int argc, char** argv) struct jbpf_io_config io_config = {}; io_config.type = JBPF_IO_IPC_PRIMARY; - std::strncpy(io_config.jbpf_path, config.run_path.c_str(), JBPF_RUN_PATH_LEN - 1); + std::strncpy(io_config.jbpf_path, config.jbpf.run_path.c_str(), JBPF_RUN_PATH_LEN - 1); io_config.jbpf_path[JBPF_RUN_PATH_LEN - 1] = '\0'; std::strncpy(io_config.jbpf_namespace, JBPF_DEFAULT_NAMESPACE, JBPF_NAMESPACE_LEN - 1); io_config.jbpf_namespace[JBPF_NAMESPACE_LEN - 1] = '\0'; std::strncpy(io_config.ipc_config.addr.jbpf_io_ipc_name, - config.ipc_name.c_str(), + config.jbpf.ipc_name.c_str(), JBPF_IO_IPC_MAX_NAMELEN - 1); io_config.ipc_config.addr.jbpf_io_ipc_name[JBPF_IO_IPC_MAX_NAMELEN - 1] = '\0'; - io_config.ipc_config.mem_cfg.memory_size = config.mem_size; + io_config.ipc_config.mem_cfg.memory_size = config.jbpf.mem_size_bytes; std::printf("[E3Controller] Initializing jbpf IO (IPC primary)...\n"); auto* io_ctx = jbpf_io_init(&io_config); @@ -509,16 +394,23 @@ int main(int argc, char** argv) // by E3SMLayer1 (RF=2). Lazy-started by libe3 on the first dApp // subscription to RF=2. e3sm_pipeline::IqPipeline::Config pipeline_cfg; - pipeline_cfg.lcm_socket_path = config.lcm_socket_path; - pipeline_cfg.codelet_base_path = config.codelet_base_path; - pipeline_cfg.expected_num_prbu = config.num_prbs; - pipeline_cfg.worker_core = config.worker_core; + pipeline_cfg.lcm_socket_path = config.jbpf.lcm_socket_path; + pipeline_cfg.codelet_base_path = config.jbpf.codelet_base_path; + pipeline_cfg.expected_num_prbu = config.radio.nof_prbs; + pipeline_cfg.worker_core = config.threads.worker_core; e3sm_pipeline::IqPipeline iq_pipeline(dispatcher, io_ctx, std::move(pipeline_cfg)); e3sm_pipeline::SlotIqPipeline::Config slot_pipeline_cfg; - slot_pipeline_cfg.lcm_socket_path = config.lcm_socket_path; - slot_pipeline_cfg.codelet_base_path = config.codelet_base_path; - slot_pipeline_cfg.worker_core = config.worker_core; + slot_pipeline_cfg.lcm_socket_path = config.jbpf.lcm_socket_path; + slot_pipeline_cfg.codelet_base_path = config.jbpf.codelet_base_path; + slot_pipeline_cfg.worker_core = config.threads.worker_core; + /* Gnb mode: the pipeline tells the codelet (and through it the gNB-side + * helper) where /e3_ran_buffers is and what shape its rows are. The + * controller owns the region, so it is the one that gets to say. */ + slot_pipeline_cfg.writer = config.shm.writer; + slot_pipeline_cfg.shm_name = config.shm.name; + slot_pipeline_cfg.radio = config.radio; + slot_pipeline_cfg.cbf16_scale = config.shm.cbf16_scale; e3sm_pipeline::SlotIqPipeline slot_iq_pipeline(dispatcher, io_ctx, std::move(slot_pipeline_cfg)); @@ -528,16 +420,20 @@ int main(int argc, char** argv) // - RF=2 L1-KPM SM: slot-level SHM-pointer IQ indications in the // configured encoding. Wired to the SlotIqPipeline. libe3::ErrorCode sm_result; - sm_result = agent.register_sm(std::make_unique(iq_pipeline, agent)); + sm_result = agent.register_sm(std::make_unique( + iq_pipeline, agent, spectrum_stats_log_path)); if (sm_result != libe3::ErrorCode::SUCCESS) { std::cerr << "Failed to register Spectrum SM (RF=1): " << libe3::error_code_to_string(sm_result) << "\n"; return 1; } - sm_result = agent.register_sm(std::make_unique( - slot_iq_pipeline, agent, - config.shm_name, config.shm_size, config.target_slot, - config.stats_log_path)); + /* Keep a borrowed pointer so the shutdown path can print the drop + * accounting. The agent owns the SM and outlives this scope's use of the + * pointer (the summary is printed before `agent` is destroyed), so this is a + * read-only borrow, not a lifetime claim. */ + auto layer1_sm = std::make_unique(slot_iq_pipeline, agent, config); + E3SMLayer1* layer1_ptr = layer1_sm.get(); + sm_result = agent.register_sm(std::move(layer1_sm)); if (sm_result != libe3::ErrorCode::SUCCESS) { std::cerr << "Failed to register L1-KPM SM (RF=2): " << libe3::error_code_to_string(sm_result) << "\n"; @@ -577,20 +473,20 @@ int main(int argc, char** argv) // libe3 spawns its IO threads inside start() and they inherit the parent's // affinity at the time of spawn, so pinning here keeps them on the unpinned // set and reserves our core for the jbpf shared-memory drain. - bool busy_poll = (config.poll_core >= 0); + bool busy_poll = (config.threads.poll_core >= 0); if (busy_poll) { cpu_set_t mask; CPU_ZERO(&mask); - CPU_SET(config.poll_core, &mask); + CPU_SET(config.threads.poll_core, &mask); if (pthread_setaffinity_np(pthread_self(), sizeof(mask), &mask) != 0) { std::fprintf(stderr, "[E3Controller] WARNING: failed to pin polling thread to core %d (%s); " "falling back to sleep-based polling\n", - config.poll_core, std::strerror(errno)); + config.threads.poll_core, std::strerror(errno)); busy_poll = false; } else { std::printf("[E3Controller] Polling thread pinned to core %d (busy-poll)\n", - config.poll_core); + config.threads.poll_core); } } @@ -602,8 +498,8 @@ int main(int argc, char** argv) #if defined(__x86_64__) || defined(__i386__) _mm_pause(); // hyperthread-friendly hint #endif - } else if (config.poll_interval_us > 0) { - usleep(config.poll_interval_us); + } else if (config.threads.poll_interval_us > 0) { + usleep(config.threads.poll_interval_us); } } @@ -620,6 +516,26 @@ int main(int argc, char** argv) iq_pipeline.stop(); std::printf("[E3Controller] Shutting down jbpf IO...\n"); jbpf_io_stop(); + + /* Drop accounting, printed AFTER both pipelines have joined their workers so + * the counters are final and no thread is still incrementing them. + * + * Two independent sources, because a slot can be lost in two places and + * conflating them would point at the wrong fix: + * + * SlotIqPipeline the SPSC queue between the jbpf dispatcher and the + * worker was full -- the controller could not keep up. + * E3SMLayer1 the slot reached the SM but did not become an + * indication, broken out by reason. + * + * Neither can see a slot the RAN never produced: under TDD only UL slots + * fire the hook, so a missing slot index is not necessarily a loss. */ + std::printf("\n[E3Controller] --- drop accounting ---\n"); + const uint64_t queue_drops = slot_iq_pipeline.dropped_samples(); + std::printf("[SlotIqPipeline] slots dropped on a full SPSC queue: %lu\n", + static_cast(queue_drops)); + std::printf("%s\n", layer1_ptr->drops_summary().c_str()); + std::printf("[E3Controller] Stopped.\n"); return 0; diff --git a/src/e3sm/asn/CMakeLists.txt b/src/e3sm/asn/CMakeLists.txt index 23aac41..3542ae2 100644 --- a/src/e3sm/asn/CMakeLists.txt +++ b/src/e3sm/asn/CMakeLists.txt @@ -89,18 +89,16 @@ set(ASN1C_SRCS "${ASN1C_OUTPUT_DIR}/Spectrum-IQDataIndication.c" "${ASN1C_OUTPUT_DIR}/Spectrum-PRBBlacklistControl.c" "${ASN1C_OUTPUT_DIR}/Spectrum-RanFunctionData.c" - "${ASN1C_OUTPUT_DIR}/Spectrum-ConfigControl.c" "${ASN1C_OUTPUT_DIR}/L1KPM-ShmRef.c" "${ASN1C_OUTPUT_DIR}/L1KPM-Indication.c" ) -# BOOLEAN: Spectrum-ConfigControl uses BOOLEAN, but we do NOT compile it here. -# libe3's installed asn1 runtime (asn1_e3ap) now ships the BOOLEAN primitive — -# build.sh stages asn1c's BOOLEAN.* skeletons into libe3's generated dir before -# building libe3 (stock libe3 0.0.4 lists them but the asn1c fork doesn't emit -# them for libe3's BOOLEAN-free E3AP grammar). So asn_DEF_BOOLEAN is resolved by -# linking libe3 (same asn1c version => ABI-compatible, single definition). The -# cleanup step drops our own BOOLEAN*.c accordingly. +# Spectrum-ConfigControl is generated (it stays in the grammar for wire +# compatibility) but deliberately NOT compiled: it is the only type that uses +# BOOLEAN, whose runtime libe3 >= 0.0.7 no longer ships (asn_DEF_BOOLEAN would +# be undefined at link), and the controller has no config-control code path. +# If a ConfigControl handler is ever added, restore the source here plus a +# BOOLEAN runtime compiled with libe3-matching ASN_DISABLE_* flags. # Static library from generated ASN.1 sources (types only, no runtime) add_library(e3sm_asn STATIC ${ASN1C_SRCS}) diff --git a/src/e3sm/asn/cleanup_asn1c.cmake b/src/e3sm/asn/cleanup_asn1c.cmake index 5a2b3d9..4f0d640 100644 --- a/src/e3sm/asn/cleanup_asn1c.cmake +++ b/src/e3sm/asn/cleanup_asn1c.cmake @@ -3,9 +3,9 @@ # with libe3's runtime (which provides the runtime via asn1_e3ap). We keep: # Spectrum-* → RF=1 SM (in-band IQ + PRB blacklist controls) # L1KPM-* → RF=2 SM (SHM-pointer layer-1 indications) -# BOOLEAN* is intentionally NOT kept: libe3's runtime now ships it (build.sh -# stages asn1c's BOOLEAN skeletons into libe3), so we link asn_DEF_BOOLEAN from -# libe3 rather than compiling a second (duplicate) copy here. +# BOOLEAN* is intentionally NOT kept: the only type that used it, +# Spectrum-ConfigControl, is generated but no longer compiled (see +# CMakeLists.txt), and libe3 >= 0.0.7 does not ship a BOOLEAN runtime either. set(ASN1C_DIR "${CMAKE_BINARY_DIR}/asn1c_generated") if(NOT EXISTS "${ASN1C_DIR}") diff --git a/src/e3sm/asn/e3sm_layer1.asn b/src/e3sm/asn/e3sm_layer1.asn index 815a3d4..15c8444 100644 --- a/src/e3sm/asn/e3sm_layer1.asn +++ b/src/e3sm/asn/e3sm_layer1.asn @@ -1,14 +1,23 @@ -- L1-KPM Service Model ASN.1 Definitions -- +-- Portions of this ASN.1 grammar are derived from NVIDIA Aerial E3 +-- message schemas. +-- +-- Original source: +-- https://github.com/NVIDIA/aerial-sample-apps (dapps/docs/e3_message_schemas.json) +-- +-- Original copyright: +-- Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +-- SPDX-License-Identifier: Apache-2.0 +-- +-- NVIDIA publishes the JSON Schema; this ASN.1 grammar is an independently +-- authored translation. Modifications and additions by Institute for Intelligent Networked +-- Systems (INSI) at Northeastern University are identified separately. +-- -- PHY-layer KPM SM at RAN Function ID 2. Telemetry-only: emits one -- indication payload per UL slot (or per period), referencing the -- /e3_ran_buffers POSIX shm region for the actual IQ bytes. -- --- ATTRIBUTION: this ASN.1 grammar is DERIVED FROM NVIDIA Aerial's public E3 --- message schema (indicationMessage.protocolData), Copyright (c) 2026 NVIDIA --- CORPORATION & AFFILIATES, SPDX-License-Identifier: Apache-2.0. See --- https://github.com/NVIDIA/aerial-sample-apps (dapps/docs/e3_message_schemas.json) --- NVIDIA publishes the JSON schema only. -- -- { -- "iq_samples": { "shm_name", "fh_buffer_index", "fh_write_index" }, diff --git a/src/e3sm/iq_pipeline.cpp b/src/e3sm/iq_pipeline.cpp index d084c45..9c0c231 100644 --- a/src/e3sm/iq_pipeline.cpp +++ b/src/e3sm/iq_pipeline.cpp @@ -165,12 +165,16 @@ libe3::ErrorCode IqPipeline::start() { } // Register the dispatcher stream LAST so we only start receiving once - // the worker is alive. Buffer release is handled by the dispatcher. + // the worker is alive. defer_release=true → we own the jbpf ring + // buffers from the moment they land here; the worker releases each + // after the consumer fan-out (zero-copy path — no 8 KB memcpy on the + // poll thread). dispatcher_.register_stream( ecpri_iq_stream_id_, [this](struct jbpf_io_stream_id* sid, void** bufs, int n) { process_buffers(sid, bufs, n); - }); + }, + /*defer_release=*/true); return libe3::ErrorCode::SUCCESS; } @@ -186,6 +190,19 @@ void IqPipeline::stop() { worker_.join(); } + // The worker may have exited (running_=false) with buffers still queued. + // We own them (defer_release=true), so release any it didn't reach BEFORE + // unload_codelets() tears down the channel/mempool below. The worker has + // joined, so this is the only thread touching the queue now. + { + const uint64_t head = queue_head_.load(std::memory_order_relaxed); + uint64_t tail = queue_tail_.load(std::memory_order_relaxed); + for (; tail != head; ++tail) { + jbpf_io_channel_release_buf(queue_[tail & (QUEUE_CAPACITY - 1)].buf); + } + queue_tail_.store(head, std::memory_order_relaxed); + } + const uint64_t dropped = dropped_.load(std::memory_order_relaxed); if (dropped > 0) { std::printf("[IqPipeline] Dropped %lu samples due to full SPSC queue\n", @@ -220,37 +237,56 @@ void IqPipeline::process_buffers(struct jbpf_io_stream_id* /*stream_id*/, } // One "now" for the whole batch — saves a syscall per buffer. + // CLOCK_REALTIME ns matches the codelet's sample.codelet_ts_ns clock + // domain so consumers can subtract them for codelet_to_dispatch_us. + struct timespec dispatch_ts; + clock_gettime(CLOCK_REALTIME, &dispatch_ts); + const uint64_t dispatch_ts_ns = + static_cast(dispatch_ts.tv_sec) * 1000000000ULL + + static_cast(dispatch_ts.tv_nsec); const uint32_t now_us = static_cast( - std::chrono::duration_cast( - std::chrono::system_clock::now().time_since_epoch()).count() - & 0x7FFFFFFFULL); + (dispatch_ts_ns / 1000ULL) & 0x7FFFFFFFULL); for (int i = 0; i < num_bufs; i++) { auto* sample = static_cast(bufs[i]); // Direction filter (1 = UL); num_prbu filter (drop SRS/PRACH/control). - if (sample->direction != 1) continue; - if (config_.expected_num_prbu > 0 - && sample->num_prbu != config_.expected_num_prbu) continue; + // defer_release=true means we own the buffer even for filter-drops - + // release it right away so the jbpf ring slot goes back into circulation. + if (sample->direction != 1 + || (config_.expected_num_prbu > 0 + && sample->num_prbu != config_.expected_num_prbu)) { + jbpf_io_channel_release_buf(bufs[i]); + continue; + } // SPSC enqueue. Single producer, so head_ is relaxed; tail_ acquire // makes the worker's progress visible to our capacity check. const uint64_t head = queue_head_.load(std::memory_order_relaxed); const uint64_t tail = queue_tail_.load(std::memory_order_acquire); if (head - tail >= QUEUE_CAPACITY) { + // Queue full: worker will never see this buffer, so release it here + // ourselves to avoid leaking the jbpf ring slot. dropped_.fetch_add(1, std::memory_order_relaxed); + jbpf_io_channel_release_buf(bufs[i]); continue; } + // Option A: enqueue the buffer POINTER (zero-copy) - no 8 KB memcpy. + // The worker reads sample fields straight from the jbpf ring and releases + // it after the consumer fan-out. QueueEntry& slot = queue_[head & (QUEUE_CAPACITY - 1)]; - std::memcpy(&slot.sample, sample, sizeof(*sample)); + slot.buf = bufs[i]; const uint32_t codelet_us = static_cast(sample->timestamp); - slot.recv_us = (codelet_us > 0) ? ((now_us - codelet_us) & 0x7FFFFFFFu) : 0; - slot.sample_id = ++total_samples_received_; + slot.recv_us = (codelet_us > 0) ? ((now_us - codelet_us) & 0x7FFFFFFFu) : 0; + slot.dispatch_ts_ns = dispatch_ts_ns; + slot.sample_id = ++total_samples_received_; queue_head_.store(head + 1, std::memory_order_release); } - // Buffer release stays with the dispatcher — do NOT release here. + // We own the buffers (defer_release=true): enqueued ones are released by the + // worker after fan-out, dropped/filtered ones were released above. So the + // dispatcher must NOT release here - and it won't (defer_release). } void IqPipeline::worker_loop() { @@ -264,7 +300,11 @@ void IqPipeline::worker_loop() { continue; } - dispatch_sample(queue_[tail & (QUEUE_CAPACITY - 1)]); + QueueEntry& entry = queue_[tail & (QUEUE_CAPACITY - 1)]; + dispatch_sample(entry); + // Consumers are synchronous, so the jbpf buffer is safe to release now. + // We own it (defer_release=true) - release exactly once per enqueued slot. + jbpf_io_channel_release_buf(entry.buf); queue_tail_.store(tail + 1, std::memory_order_release); } @@ -272,7 +312,9 @@ void IqPipeline::worker_loop() { void IqPipeline::dispatch_sample(const QueueEntry& entry) { using clock = std::chrono::steady_clock; - const auto& sample = entry.sample; + // entry.buf points straight at the jbpf ring buffer (Option A, zero-copy). + // Valid until the worker releases it right after this fan-out returns. + const auto& sample = *static_cast(entry.buf); const auto t_decompress_start = clock::now(); if (!e3sm_spectrum::decompress_bfp_9bit( @@ -292,6 +334,9 @@ void IqPipeline::dispatch_sample(const QueueEntry& entry) { dec.decompressed_size = decompressed_buf_.size(); dec.recv_us = entry.recv_us; dec.sample_id = entry.sample_id; + dec.gnb_ts_ns = sample.gnb_ts_ns; + dec.codelet_ts_ns = sample.codelet_ts_ns; + dec.dispatch_ts_ns = entry.dispatch_ts_ns; dec.decompress_ns = std::chrono::duration_cast( t_decompress_end - t_decompress_start).count(); diff --git a/src/e3sm/iq_pipeline.h b/src/e3sm/iq_pipeline.h index 04a252b..b630a1d 100644 --- a/src/e3sm/iq_pipeline.h +++ b/src/e3sm/iq_pipeline.h @@ -61,6 +61,18 @@ struct DecompressedSample { uint32_t recv_us{0}; uint32_t sample_id{0}; + // RAN-side stage timestamps (CLOCK_REALTIME ns) forwarded verbatim + // from the codelet's iq_sample_data. Consumers subtract these to + // report the same three-stage schema as the SlotIqPipeline + // (gnb_to_codelet_us / codelet_to_dispatch_us / dispatch_to_handler_us). + // gnb_ts_ns — stamped at hook fire in ofh_message_receiver_impl + // codelet_ts_ns — stamped by the codelet right before ringbuf output + // dispatch_ts_ns — stamped by IqPipeline::process_buffers on the + // dispatcher poll thread (batch-shared) + uint64_t gnb_ts_ns{0}; + uint64_t codelet_ts_ns{0}; + uint64_t dispatch_ts_ns{0}; + // BFP decompression duration in nanoseconds — used by SMs to log // per-slot data-plane cost in their statistics_*.log files. uint64_t decompress_ns{0}; @@ -108,10 +120,25 @@ class IqPipeline { uint64_t dropped_samples() const { return dropped_.load(std::memory_order_relaxed); } private: - static constexpr std::size_t QUEUE_CAPACITY = 256; // power of 2 + // At most the codelet's jbpf ring depth (64, see ecpri_iq_collect.c + // `jbpf_ringbuf_map(output_map, ..., 64)`) can be un-released at once, + // so a matching queue never needs more entries. Power of 2 for cheap + // masking; larger than the ring is dead weight. + static constexpr std::size_t QUEUE_CAPACITY = 64; struct QueueEntry { - struct iq_sample_data sample; + // Pointer to the jbpf ring buffer (an iq_sample_data) the codelet + // wrote. Option A: enqueue the POINTER (zero-copy) instead of + // memcpy'ing the ~8 KB iq_sample_data on the poll thread. The + // dispatcher stream is registered with defer_release=true, so the + // dispatcher does NOT free this buffer; the worker releases it + // (jbpf_io_channel_release_buf) after the consumer fan-out. + // Valid from enqueue until that release. + void* buf; + // Dispatcher-thread stamp (CLOCK_REALTIME ns), shared across the + // whole poll batch. Consumers subtract sample.codelet_ts_ns from + // this to get codelet_to_dispatch_us, matching SlotIqPipeline. + uint64_t dispatch_ts_ns; uint32_t recv_us; uint32_t sample_id; }; diff --git a/src/e3sm/l1_kpm/e3sm_layer_1.cpp b/src/e3sm/l1_kpm/e3sm_layer_1.cpp index 421189a..36fb79b 100644 --- a/src/e3sm/l1_kpm/e3sm_layer_1.cpp +++ b/src/e3sm/l1_kpm/e3sm_layer_1.cpp @@ -14,14 +14,146 @@ #include "e3sm_layer1_wrapper.h" #include "e3sm_layer1_json.h" +#include "l1_kpm_trace.h" #include #include -#include #include +#include + +namespace trace = e3sm_l1kpm_trace; + +/* --------------------------------------------------------------------------- + * Drop accounting + * + * The point of naming every reason separately: "we dropped 4000 slots" is not + * actionable, whereas "4000 NoSubscribers" (nobody had subscribed yet) and + * "4000 RanPublishedNothing" (the gNB helper never wrote a row) call for + * completely different fixes, and one of them is not even a fault. + * ------------------------------------------------------------------------- */ + +const char* E3SMLayer1::drop_name(Drop d) { + switch (d) { + case Drop::NotRunning: return "not_running"; + case Drop::SlotFiltered: return "slot_filtered"; + case Drop::GeometryMismatch: return "geometry_mismatch"; + case Drop::NoIq: return "no_iq"; + case Drop::BlobTooSmall: return "blob_too_small"; + case Drop::RanPublishedNothing: return "ran_published_nothing"; + case Drop::NoSubscribers: return "no_subscribers"; + case Drop::EncodeFailed: return "encode_failed"; + case Drop::EmitFailed: return "emit_failed"; + case Drop::COUNT: break; + } + return "unknown"; +} + +uint64_t E3SMLayer1::total_drops() const { + uint64_t t = 0; + for (std::size_t i = 0; i < kDropCount; ++i) { + t += drops_[i].load(std::memory_order_relaxed); + } + return t; +} + +void E3SMLayer1::note_drop(Drop d) { + drops_[static_cast(d)].fetch_add(1, std::memory_order_relaxed); + maybe_log_drops(); +} + +void E3SMLayer1::maybe_log_drops() { + /* Throttle to <=1/s. Without this, a condition that drops EVERY slot -- no + * subscriber yet, or a geometry mismatch -- would emit a line and a CSV row + * at slot rate (2000/s), which is both useless and a real cost on the + * worker thread. */ + const auto now = std::chrono::steady_clock::now(); + if (drops_last_report_.time_since_epoch().count() != 0 && + now - drops_last_report_ < std::chrono::seconds(1)) { + return; + } + drops_last_report_ = now; + + const uint64_t total = total_drops(); + if (total == drops_last_total_) { + return; /* nothing new since the last report */ + } + const uint64_t since = total - drops_last_total_; + drops_last_total_ = total; + + /* Live line, so a run that is silently dropping everything is visible + * without waiting for shutdown or opening the CSV. */ + std::fprintf(stderr, + "[E3SMLayer1] dropped %lu slot(s) in the last second (%lu total, %lu published)\n", + static_cast(since), + static_cast(total), + static_cast(published_slots())); + + if (drops_log_path_.empty()) { + return; /* counters still maintained; only the CSV is opt-in */ + } + if (!drops_log_.is_open()) { + drops_log_.open(drops_log_path_, std::ios::out | std::ios::trunc); + if (!drops_log_.is_open()) { + std::fprintf(stderr, "[E3SMLayer1] cannot open drop log %s\n", + drops_log_path_.c_str()); + drops_log_path_.clear(); /* do not retry every second */ + return; + } + drops_log_ << "uptime_s,published,dropped_total,latrec_clamped"; + for (std::size_t i = 0; i < kDropCount; ++i) { + drops_log_ << ',' << drop_name(static_cast(i)); + } + drops_log_ << '\n'; + drops_log_start_ = now; + } + + const double uptime_s = + std::chrono::duration(now - drops_log_start_).count(); + drops_log_ << uptime_s << ',' << published_slots() << ',' << total + << ',' << trace::clamped(); + for (std::size_t i = 0; i < kDropCount; ++i) { + drops_log_ << ',' << drops_[i].load(std::memory_order_relaxed); + } + drops_log_ << '\n'; + /* Aggregate and throttled to <=1/s, so this flush is nowhere near the slot + * path -- unlike the per-slot stage CSV this replaced, which flushed once + * per slot from inside the handler. */ + drops_log_.flush(); /* a crash mid-run must not lose the accounting */ +} + +std::string E3SMLayer1::drops_summary() const { + const uint64_t total = total_drops(); + const uint64_t pub = published_slots(); + + std::ostringstream os; + os << "[E3SMLayer1] slots published: " << pub << ", dropped: " << total; + /* Capture quality, not a drop: a non-zero count means the ring's ascending + * invariant had to be enforced on that many stamps, so A1 is understated + * for those slots. Worth seeing even on a run that dropped nothing. */ + if (const uint64_t clamped = trace::clamped(); clamped != 0) { + os << "\n latrec stamps clamped: " << clamped + << " (A1 understated for these; see l1_kpm_trace.h)"; + } + if (total == 0) { + os << " (none)"; + return os.str(); + } + /* Share of everything that reached this SM, which is the number worth + * quoting: dropped/(published+dropped), not dropped/published. */ + const double pct = 100.0 * static_cast(total) / + static_cast(total + pub); + os << " (" << pct << "% of arrivals)"; + for (std::size_t i = 0; i < kDropCount; ++i) { + const uint64_t c = drops_[i].load(std::memory_order_relaxed); + if (c != 0) { + os << "\n " << drop_name(static_cast(i)) << ": " << c; + } + } + return os.str(); +} libe3::ErrorCode E3SMLayer1::init() { - if (!shm_writer_.open(shm_name_, shm_size_)) { + if (!shm_writer_.open(shm_name_, shm_size_, geom_, cbf16_scale_, writer_mode_)) { return libe3::ErrorCode::INTERNAL_ERROR; } @@ -111,68 +243,121 @@ std::vector E3SMLayer1::ran_function_data() const { } void E3SMLayer1::on_sample(const e3sm_pipeline::SlotSample& s) { - if (!running_) return; + if (!running_) { note_drop(Drop::NotRunning); return; } // Optional single-slot filter (debug/diagnostic): target_slot_ is // the absolute slot index within a radio frame (subframe_id*2 + // slot_id under 30 kHz SCS). -1 disables. if (target_slot_ >= 0) { - const int abs_slot = - static_cast(s.subframe_id) * 2 + static_cast(s.slot_id); - if (abs_slot != target_slot_) return; + // slot_id is already ocudu's slot_point::slot_index(), i.e. the index + // within the 10 ms radio frame (0..slots_per_frame()-1), so use it + // directly rather than recomputing from subframe_id with a hardcoded + // 2 slots/subframe that only held at 30 kHz SCS. + const int abs_slot = static_cast(s.slot_id); + if (abs_slot != target_slot_) { note_drop(Drop::SlotFiltered); return; } } - // Sanity: the blob must carry every antenna port we intend to publish - // (nof_ports × kShmSymbolsPerRow × kShmScPerSymbol × 4 bytes/cbf16), - // clamped to the SHM row's antenna capacity (kShmAntsLayout). Catches a - // grid-shape drift (e.g. ocudu shipping fewer ports than nof_ports claims). - const uint16_t ports_clamped = - (s.nof_ports == 0) ? 1u - : (s.nof_ports > e3sm_spectrum::kShmAntsLayout - ? static_cast(e3sm_spectrum::kShmAntsLayout) - : s.nof_ports); - const uint32_t kPerAntBytes = - static_cast(e3sm_spectrum::kShmSymbolsPerRow) * - static_cast(e3sm_spectrum::kShmScPerSymbol) * 4u; - const uint32_t kExpectedBytes = static_cast(ports_clamped) * kPerAntBytes; - if (s.iq == nullptr || s.iq_size_bytes < kExpectedBytes) { - std::fprintf(stderr, - "[E3SMLayer1] Slot blob too small: %u bytes (need >= %u). " - "Grid shape mismatch (nof_ports=%u nof_symbols=%u nof_subc=%u)?\n", - s.iq_size_bytes, kExpectedBytes, - s.nof_ports, s.nof_symbols, s.nof_subcarriers); + // Validate the CONFIGURED geometry against what the RAN actually reports, + // once, on the first slot. + if (!geometry_checked_) { + geometry_checked_ = true; + std::string gerr; + if (!e3config::validate_against_ran(geom_, s.nof_ports, s.nof_symbols, s.nof_subcarriers, gerr)) { + geometry_ok_ = false; + std::fprintf(stderr, + "[E3SMLayer1] GEOMETRY MISMATCH — refusing to publish.\n" + " configured: %s\n" + " RAN slot: %u ports x %u sym x %u subc\n" + " %s\n", + geom_.describe().c_str(), s.nof_ports, s.nof_symbols, s.nof_subcarriers, gerr.c_str()); + } else { + std::printf("[E3SMLayer1] geometry confirmed against the RAN: %s\n", geom_.describe().c_str()); + } + } + if (!geometry_ok_) { + note_drop(Drop::GeometryMismatch); return; } - using clock = std::chrono::steady_clock; - - // CLOCK_REALTIME ns at on_sample entry. Same domain as - // s.gnb_ts_ns / s.codelet_ts_ns / s.dispatch_ts_ns, so the four - // RAN-side stage durations subtract cleanly into statistics_layer1.log. - auto realtime_ns_now = []() -> uint64_t { - struct timespec ts; - clock_gettime(CLOCK_REALTIME, &ts); - return static_cast(ts.tv_sec) * 1000000000ULL + - static_cast(ts.tv_nsec); - }; - const uint64_t handler_entry_ns = realtime_ns_now(); + // Sanity: the blob must carry every antenna port we intend to publish, + // clamped to the SHM row's antenna capacity. Catches a grid-shape drift + // (e.g. ocudu shipping fewer ports than nof_ports claims). + const uint16_t ports_clamped = + (s.nof_ports == 0) ? 1u + : (s.nof_ports > geom_.nof_ports ? geom_.nof_ports : s.nof_ports); + if (shm_writer_.writes_rows()) { + /* Controller mode: the blob must carry every antenna port we intend to + * publish. Catches a grid-shape drift (e.g. ocudu shipping fewer ports + * than nof_ports claims). */ + const uint32_t kPerAntBytes = geom_.cbf16_bytes_per_ant(); + const uint32_t kExpectedBytes = static_cast(ports_clamped) * kPerAntBytes; + if (s.iq == nullptr || s.iq_size_bytes == 0) { + /* Not a grid-shape problem: zero bytes means the codelet published a + * DESCRIPTOR, i.e. it ran the gNB-side publish path, while this + * controller is configured to convert the IQ itself. The shipped + * codelet only does the descriptor path, so `shm.writer: controller` + * cannot work with it. Warn once — at slot rate this would otherwise + * bury the log. */ + if (!writer_mode_warned_) { + writer_mode_warned_ = true; + std::fprintf(stderr, + "[E3SMLayer1] shm.writer is 'controller' but the codelet published a " + "descriptor with no IQ (0 bytes).\n" + " The shipped uplink_slot_samples codelet writes rows via the gNB-side " + "helper and sends only a ~64 B descriptor,\n" + " so the controller has nothing to convert. Set shm.writer: gnb in the " + "config.\n" + " (Grid dims reported by the RAN are fine: %u ports x %u sym x %u subc.)\n", + s.nof_ports, s.nof_symbols, s.nof_subcarriers); + } + note_drop(Drop::NoIq); + return; + } + if (s.iq_size_bytes < kExpectedBytes) { + std::fprintf(stderr, + "[E3SMLayer1] Slot blob too small: %u bytes (need >= %u). " + "Grid shape mismatch (nof_ports=%u nof_symbols=%u nof_subc=%u)?\n", + s.iq_size_bytes, kExpectedBytes, + s.nof_ports, s.nof_symbols, s.nof_subcarriers); + note_drop(Drop::BlobTooSmall); + return; + } + } else { + /* gNB mode: no IQ crosses the jbpf ring — the helper already wrote the + * row. What can go wrong here is the helper REFUSING, which it signals + * with TRUNCATED and bytes_written == 0 rather than writing a partial + * row. Publishing an Indication for that would point the dApp at a row + * nobody wrote. */ + if ((s.flags & E3_SLOT_FLAG_TRUNCATED) != 0 || s.bytes_written == 0) { + if (!truncated_warned_) { + truncated_warned_ = true; + std::fprintf(stderr, + "[E3SMLayer1] gNB helper published nothing for this slot " + "(flags=0x%02x bytes_written=%u, grid %u ports x %u sym x %u subc). " + "Check that the region geometry matches and that the helper attached.\n", + s.flags, s.bytes_written, s.nof_ports, s.nof_symbols, s.nof_subcarriers); + } + note_drop(Drop::RanPublishedNothing); + return; + } + } // Single encoding: one wire format for every subscriber. const libe3::EncodingFormat enc = agent_ ? agent_->config().encoding : libe3::EncodingFormat::ASN1; const auto subs = get_subscribers(); if (subs.empty()) { + /* Counted, but NOT a fault: before the first dApp subscribes every slot + * lands here, so a large no_subscribers count on a healthy run is + * normal and is exactly why the summary breaks reasons out. + * + * Checked before the first stage stamp deliberately: this is the SM's + * normal idle state, and stamping it would fill the ring with records + * for slots nobody asked for. */ + note_drop(Drop::NoSubscribers); return; // No one is listening; nothing to publish. } - // Publish the slot to /e3_ran_buffers. publish_row_cbf16 does the - // bf16 → fp16 conversion inline (the dApp expects fp16 on wire). - const auto t_shm_start = clock::now(); - uint8_t fh_buf_idx = 0; - uint32_t fh_write_idx = 0; - shm_writer_.publish_row_cbf16(s.iq, ports_clamped, fh_buf_idx, fh_write_idx); - const auto t_shm_end = clock::now(); - // Absolute slot within a 10 ms frame (0..19 at 30 kHz SCS). // // SlotIqPipeline forwards ocudu's slot_point::slot_index() directly @@ -187,10 +372,41 @@ void E3SMLayer1::on_sample(const e3sm_pipeline::SlotSample& s) { // 19 becomes 37 (both invalid). const uint16_t abs_slot = s.slot_id; - // Encode the slot once in the configured wire format. Encoder is + // Everything from here to emit_tail() is one row in the stage records. + // publish_seq keys it, including inside libe3 (see trace::bind_libe3). + const uint64_t publish_seq = + slot_publish_seq_.fetch_add(1, std::memory_order_relaxed) + 1; + + // A1: the data recording, replayed from the two boundaries the slot carries. + // Both stamps are back-dated to when they actually happened - the gNB, the + // codelet and the dispatcher all read CLOCK_MONOTONIC, which is latrec's own + // clock, so there is no conversion and no offset arithmetic. + trace::record_begin(publish_seq, s.sfn, abs_slot, s.gnb_ts_ns, s.codelet_entry_ts_ns); + trace::process_begin(publish_seq, s.codelet_ts_ns, s.dispatch_ts_ns, + (s.bytes_written > 0) ? s.bytes_written : s.iq_size_bytes); + + // Publish the slot to /e3_ran_buffers. + // + // In `writer: controller` mode this converts cbf16 -> fp16 inline (the dApp + // expects fp16 on the wire), and that cost lands inside A2. In `writer: gnb` + // mode the gNB-side helper has already written the row in the PHY RX thread + // and reported which one - that cost is inside A1 - so we must NOT write + // here: both writers keep their own ring cursor and two active writers would + // silently overwrite each other. publish_row_cbf16 hard-refuses in that mode; + // we take the indices the RAN reported instead. + uint8_t fh_buf_idx = 0; + uint32_t fh_write_idx = 0; + if (shm_writer_.writes_rows()) { + shm_writer_.publish_row_cbf16(s.iq, ports_clamped, fh_buf_idx, fh_write_idx); + } else { + fh_buf_idx = s.fh_buffer_index; + fh_write_idx = s.fh_write_index; + } + + // A3: encode the slot once in the configured wire format. Encoder is // O(small); we do it once per slot regardless of subscriber count. const bool want_json = (enc == libe3::EncodingFormat::JSON); - const auto t_encode_start = clock::now(); + trace::encode_begin(publish_seq, s.iq_size_bytes); encoded_buf_.clear(); const bool encoded_ok = want_json ? e3sm_layer1::encode_iq_indication_json( @@ -199,21 +415,25 @@ void E3SMLayer1::on_sample(const e3sm_pipeline::SlotSample& s) { : e3sm_layer1::encode_iq_indication_aper( s.gnb_ts_ns, s.sfn, abs_slot, shm_name_, fh_buf_idx, fh_write_idx, ports_clamped, encoded_buf_); - const uint64_t encode_ns = std::chrono::duration_cast( - clock::now() - t_encode_start).count(); if (!encoded_ok) { std::fprintf(stderr, "[E3SMLayer1] Failed to %s-encode indication (buf=%u row=%u)\n", want_json ? "JSON" : "APER", fh_buf_idx, fh_write_idx); + trace::encode_failed(publish_seq); + note_drop(Drop::EncodeFailed); return; } + trace::encode_done(publish_seq, encoded_buf_.size()); - // Fan out one indication per subscriber. emit_ns here captures the - // SM-side enqueue cost (PDU build + emit_outbound). The encode + ZMQ - // send stages downstream are measured library-side in libe3's - // publisher-stage CSV (--pub-stages-log), joinable by message_id. - const auto t_emit_start = clock::now(); + // Hand off to libe3. Publishing publish_seq first is what lets the library's + // own records - E3AP encode, queuing, the connector send - be attributed back + // to this slot: libe3 stamps it into EMIT_ENTER's aux, which surfaces offline + // as the outbound leg's origin_seq. Everything downstream of here is the + // library's box, measured in the library; we do not re-time it. + trace::bind_libe3(publish_seq); + + // Fan out one indication per subscriber. for (uint32_t dapp_id : subs) { libe3::Pdu pdu = make_indication_pdu(dapp_id, RAN_FUNCTION_ID, encoded_buf_); @@ -222,52 +442,15 @@ void E3SMLayer1::on_sample(const e3sm_pipeline::SlotSample& s) { std::fprintf(stderr, "[E3SMLayer1] Failed to send indication to dApp %u: %s\n", dapp_id, libe3::error_code_to_string(rc)); + /* Per-DAPP, so with several subscribers one slot can bump this more + * than once while still being published to the others. That makes + * emit_failed the one counter which is not mutually exclusive with + * a successful publish -- deliberate, since the alternative (drop + * the slot from the count entirely) would hide a partial fan-out. */ + note_drop(Drop::EmitFailed); } } - const auto t_emit_end = clock::now(); - // --- Per-slot statistics (disabled unless --stats-log was given) --- - ++slot_publish_seq_; - if (stats_log_path_.empty()) { - return; - } - // Schema: - // slot_seq, - // gnb_to_codelet_us, ocudu hook -> codelet entry (jbpf invocation) - // codelet_to_dispatch_us,codelet -> controller dispatcher poll - // dispatch_to_handler_us,dispatcher -> on_sample (SPSC queue wait) - // shm_ns, publish_row_cbf16 cost - // encode_ns, SM payload encoder cost - // emit_ns, emit_outbound fan-out (SM-side enqueue) cost - // nof_subc, iq_bytes slot size (handy if BWP changes) - // - // The three "us" stages are derived from RAN-side CLOCK_REALTIME ns - // stamps (gNB -> codelet -> dispatcher -> handler). The post-encode - // ZMQ-send stage lives in libe3's publisher-stage CSV, joinable by - // message_id. - if (!stats_log_.is_open()) { - stats_log_.open(stats_log_path_, std::ios::out | std::ios::trunc); - stats_log_ << "slot_seq," - "gnb_to_codelet_us,codelet_to_dispatch_us,dispatch_to_handler_us," - "shm_ns,encode_ns,emit_ns,nof_subc,iq_bytes\n"; - } - auto ns_between = [](auto a, auto b) { - return std::chrono::duration_cast(b - a).count(); - }; - // Saturating subtractions: the three RAN-side stages are derived - // from absolute realtime stamps. In the (rare) clock-warp case - // they could go negative — clamp to 0 so the CSV row stays parseable. - auto sat_us = [](uint64_t lhs, uint64_t rhs) -> uint64_t { - return (lhs > rhs) ? ((lhs - rhs) / 1000ULL) : 0ULL; - }; - stats_log_ << slot_publish_seq_ << ',' - << sat_us(s.codelet_ts_ns, s.gnb_ts_ns) << ',' - << sat_us(s.dispatch_ts_ns, s.codelet_ts_ns) << ',' - << sat_us(handler_entry_ns, s.dispatch_ts_ns) << ',' - << ns_between(t_shm_start, t_shm_end) << ',' - << encode_ns << ',' - << ns_between(t_emit_start, t_emit_end) << ',' - << s.nof_subcarriers << ',' - << s.iq_size_bytes << '\n'; - stats_log_.flush(); + // The emit tail, over all subscribers. + trace::emit_tail(publish_seq, subs.size()); } diff --git a/src/e3sm/l1_kpm/e3sm_layer_1.h b/src/e3sm/l1_kpm/e3sm_layer_1.h index 0e8fe2f..1df9edd 100644 --- a/src/e3sm/l1_kpm/e3sm_layer_1.h +++ b/src/e3sm/l1_kpm/e3sm_layer_1.h @@ -27,6 +27,9 @@ #include #include +#include +#include +#include #include #include #include @@ -38,16 +41,16 @@ class E3SMLayer1 : public libe3::ServiceModel { E3SMLayer1(e3sm_pipeline::SlotIqPipeline& pipeline, libe3::E3Agent& agent, - std::string shm_name = "/e3_ran_buffers", - std::size_t shm_size = static_cast(1) << 30, - int target_slot = -1, - std::string stats_log_path = "") + const e3config::ControllerConfig& cfg) : pipeline_(pipeline), agent_(&agent), - shm_name_(std::move(shm_name)), - shm_size_(shm_size), - target_slot_(target_slot), - stats_log_path_(std::move(stats_log_path)) + shm_name_(cfg.shm.name), + shm_size_(cfg.shm.size_bytes), + target_slot_(cfg.target_slot), + drops_log_path_(cfg.logging.drops_log_path), + geom_(cfg.radio), + cbf16_scale_(cfg.shm.cbf16_scale), + writer_mode_(cfg.shm.writer) {} std::string name() const override { return "L1 KPM Service Model"; } @@ -70,7 +73,49 @@ class E3SMLayer1 : public libe3::ServiceModel { uint32_t request_message_id, const libe3::DAppControlAction& action) override; + enum class Drop : uint8_t { + NotRunning = 0, /* stopped, or a straggler after stop() */ + SlotFiltered, /* target_slot filter excluded it (deliberate) */ + GeometryMismatch, /* config vs RAN disagreement; latched refusal */ + NoIq, /* controller mode, but codelet sent descriptor */ + BlobTooSmall, /* short blob; a partial row would be wrong data */ + RanPublishedNothing, /* gNB helper flagged TRUNCATED / wrote 0 bytes */ + NoSubscribers, /* nobody subscribed to RF=2 yet */ + EncodeFailed, /* APER/JSON encode error */ + EmitFailed, /* libe3 rejected the outbound enqueue */ + COUNT + }; + static constexpr std::size_t kDropCount = static_cast(Drop::COUNT); + static const char* drop_name(Drop d); + + /* Relaxed loads: on_sample writes these on the worker thread while the + * summary reads them from main.*/ + uint64_t drop_count(Drop d) const { + return drops_[static_cast(d)].load(std::memory_order_relaxed); + } + uint64_t total_drops() const; + uint64_t published_slots() const { + return slot_publish_seq_.load(std::memory_order_relaxed); + } + + /* Multi-line human summary: published, total dropped, and each non-zero + * reason. Returns the "nothing dropped" line when all counters are zero, + * so the caller can print it unconditionally. */ + std::string drops_summary() const; + private: + /* Bump a reason. Also drives the throttled live log line. */ + void note_drop(Drop d); + + /* Append a cumulative row to drops_log_path_, at most once a second. + * Cumulative rather than per-interval so a row is meaningful on its own and + * a missed flush cannot lose events. No-op when the path is empty. + * + * This is the only file this SM writes. It is aggregate and throttled, so + * unlike the per-slot stage CSV it replaced it does no work on the slot + * path -- stage timing is latrec's job now, see l1_kpm_trace.h. */ + void maybe_log_drops(); + /* SlotIqPipeline consumer callback (worker thread). Receives one * fully-assembled UL slot from ocudu's resource grid. */ void on_sample(const e3sm_pipeline::SlotSample& s); @@ -80,21 +125,58 @@ class E3SMLayer1 : public libe3::ServiceModel { std::string shm_name_; std::size_t shm_size_; - /* Optional single-slot filter; absolute slot within a 10 ms frame - * (subframe_id*2 + slot_id at 30 kHz SCS). -1 disables. */ + /* Optional single-slot filter; slot index within a 10 ms frame, i.e. the + * value ocudu's slot_point::slot_index() returns (0..19 at 30 kHz SCS). + * -1 disables. Superseded by workstream F's codelet-side slot_mask, which + * makes the same decision before any data moves. */ int target_slot_; - /* Path for the per-slot stage CSV. Empty disables stats logging. */ - std::string stats_log_path_; + /* Path for the throttled drop-accounting CSV. Empty disables it. */ + std::string drops_log_path_; e3sm_spectrum::ShmIqWriter shm_writer_; bool running_{false}; - /* Per-slot stats. slot_publish_seq_ increments for every slot - * we publish; counts indications emitted to dApps. stats_log_ is - * opened lazily on the first published slot when stats_log_path_ - * is non-empty. */ - uint64_t slot_publish_seq_{0}; - std::ofstream stats_log_; + /* Row geometry from the YAML config; replaces the old constexpr in + * e3sm_shm_writer.h. Treated as a bootstrap and checked against the RAN on + * the first slot -- see the validate_against_ran() call in on_sample. */ + e3config::RadioGeometry geom_; + float cbf16_scale_{1.0f}; + + /* Who converts and writes the fp16 rows. Exactly one process may; see + * e3config::ShmWriter. */ + e3config::ShmWriter writer_mode_{e3config::ShmWriter::Controller}; + + /* One-shot geometry check. geometry_ok_ latches false on mismatch so we + * refuse to publish rather than emit rows the dApp would misread. */ + bool geometry_checked_{false}; + bool geometry_ok_{true}; + + /* One-shot warning when the gNB helper reports it published nothing, so a + * persistent misconfiguration does not flood the log at slot rate. */ + bool truncated_warned_{false}; + + /* One-shot: controller mode configured but the codelet only sends descriptors. */ + bool writer_mode_warned_{false}; + + /* Increments for every slot that reaches the traced region, and keys that + * slot's stage records across the whole E3 path including libe3's own -- it + * is published to the library with latrec_ctx_set() before emitting and + * comes back as the outbound leg's origin_seq. Only meaningful within this + * ring; every producer numbers from 1. + * + * Atomic because the shutdown summary reads it from the main thread while + * on_sample increments it on the worker. */ + std::atomic slot_publish_seq_{0}; + + /* Drop accounting -- see enum Drop. */ + std::atomic drops_[kDropCount]{}; + std::ofstream drops_log_; + /* Throttles both the live stderr line and the CSV row to <=1/s, so a + * pathological run cannot turn drop reporting into the bottleneck. */ + std::chrono::steady_clock::time_point drops_last_report_{}; + uint64_t drops_last_total_{0}; + /* Set when the drop CSV is opened, so uptime_s starts at 0 in the file. */ + std::chrono::steady_clock::time_point drops_log_start_{}; /* Encoded indication payload (one encoding per the agent config). * Capacity grows on first use, reused across slots. */ diff --git a/src/e3sm/l1_kpm/l1_kpm_trace.h b/src/e3sm/l1_kpm/l1_kpm_trace.h new file mode 100644 index 0000000..7d65dd5 --- /dev/null +++ b/src/e3sm/l1_kpm/l1_kpm_trace.h @@ -0,0 +1,245 @@ +/* + * Stage stamps for the L1-KPM slot path. + * + * This header owns exactly one thing: how this repository's per-slot path maps + * onto libe3's shared stage catalog (libe3/latrec.h). It is the whole of the + * controller's instrumentation - E3SMLayer1::on_sample calls these and nothing + * else, and the recording bench links this same header, so what CI measures is + * the code that runs on the radio rather than a copy of it. + * + * The catalog names *operations*, not components: the same identifier is used + * wherever that operation is performed, and which component performed it is + * read off the ring that recorded it. So there is no E3Controller-specific + * stage block to use, and inventing one would break every offline tool. + * + * Boxes covered (libe3 docs/path-a-e3-loop.md numbering), all on the forward + * leg: + * + * A1 E3 data recording RECORD_BEGIN -> PROCESS_BEGIN + * A2 Processing PROCESS_BEGIN -> ENCODE_E3SM_BEGIN + * A3 Encode E3SM ENCODE_E3SM_BEGIN -> ENCODE_E3SM_DONE + * - emit tail ENCODE_E3SM_DONE -> WAIT_ENTER + * + * A1 and A2 are recorded HERE, from the timestamps the slot carries, rather + * than by the gNB stamping its own ring. The gNB needs no instrumentation for + * this: gnb_ts_ns marks A1's entry and codelet_ts_ns marks its exit, so both + * boundaries are already on the wire by the time the slot reaches us. + * + * A1 = gnb_ts_ns -> codelet_ts_ns the data recording itself: jbpf + * dispatch, then (writer: gnb) the + * cbf16 -> fp16 convert of the whole grid + * and the row write into /e3_ran_buffers. + * A2 = codelet_ts_ns -> encode getting it to the Service Model: the + * jbpf ring transit, the dispatcher poll + * and the queue wait. + * + * A4 onwards (E3AP encode, queuing, delivery) belong to libe3 and are stamped + * inside the library. E3SM is this repository's box; E3AP is the library's. We + * do not re-time the library's stages, we join to them - see bind_libe3(). + * + * Everything here compiles to nothing when libe3 was built without + * -DLIBE3_ENABLE_LATREC: latrec.h supplies inline no-op stubs, and because + * LIBE3_ENABLE_LATREC is a PUBLIC compile definition on libe3::libe3 there is + * no way for this translation unit to disagree with the library it links. + */ +#ifndef E3_SM_L1KPM_TRACE_H +#define E3_SM_L1KPM_TRACE_H + +#include +#include + +#include +#include +#include + +namespace e3sm_l1kpm_trace { + +/* Ring role for the thread that runs the slot path. + * + * The role is the component's identity: libe3's tools/latrec2csv.py maps the + * role's first segment through its RING_COMPONENTS table to decide which + * per-component CSV the records land in, and `e3controller` is the prefix + * reserved for us (it files under ocudu.csv, alongside the jbpf/gNB capture + * side of the same deployment). Do NOT shorten this to "l1_kpm": that prefix + * is claimed by OAI and the records would be filed as OAI's. A role the table + * does not recognise lands in other.csv. + */ +inline constexpr const char* kRingRole = "e3controller.l1_kpm"; + +/* Open the calling thread's ring. Idempotent, first-open-wins, and it faults + * in the whole mapping - so call it once at the top of the thread's loop, off + * the slot path, and never from on_sample. + * + * Ordering matters: libe3's queue_outbound opens a ring on demand for whatever + * thread emits, as libe3.outbound. Since the slot path emits from this same + * thread, if libe3 got there first the ring would carry libe3's name and every + * record we stamp would be filed as the library's. Opening here, before the + * first emit, is what keeps the component attribution correct. + */ +inline void open_ring() +{ + latrec_tls_open_as(kRingRole); +} + +#ifdef LIBE3_ENABLE_LATREC + +/* Monotone floor for this thread's ring, and how often we had to enforce it. + * + * A1's boundaries happened before we were called, so RECORD_BEGIN and + * PROCESS_BEGIN are stamped with times that precede the moment we stamp them. + * That is safe here in a way it would not have been before: the gNB hook, the + * codelet and the dispatcher poll all read CLOCK_MONOTONIC now, which is + * latrec's own clock, so there is no domain conversion and no rate skew to + * absorb. + * + * What still has to hold is the ring's own invariant. A ring is a single-writer + * log whose t_ns must ascend, and libe3's reader treats the ONE permitted + * descent as the wrap point and silently rotates there - so a descent would not + * produce an error, it would produce a plausible-looking capture cut at the + * wrong offset. Ordering across slots is a program-order property and normally + * has a wide margin: the jbpf hook is a synchronous inline call, so slot N's + * codelet has returned before slot N+1 is stamped, and A1 is tens of + * microseconds against a slot spacing of at least 500 us at 30 kHz SCS. + * + * Two things can still break it, so we enforce rather than assume: + * - a pipeline stall longer than the slot spacing, and + * - several RU receive threads (one per sector) feeding one worker, whose + * slots interleave on this one ring with no ordering between them. + * + * Enforcing costs a compare and a store. A non-zero clamped() means A1 is + * understated for that many slots, which is why the count is surfaced rather + * than left to be inferred from the data. + */ +inline thread_local uint64_t tls_floor_ns = 0; +/* Process-wide, so the shutdown summary can read it from the main thread while + * the worker maintains it. Touched only when a clamp actually happens, which is + * the rare case - the floor itself is thread-local and costs a compare. */ +inline std::atomic g_clamped{0}; + +inline uint64_t monotone(uint64_t t_ns) +{ + if (t_ns <= tls_floor_ns) { + g_clamped.fetch_add(1, std::memory_order_relaxed); + t_ns = tls_floor_ns + 1; + } + tls_floor_ns = t_ns; + return t_ns; +} + +inline void stamp_at(uint64_t seq, uint8_t stage, uint64_t aux, uint64_t aux2, uint64_t t_ns) +{ + latrec_tstamp_at(seq, stage, aux, aux2, monotone(t_ns)); +} + +inline void stamp_now(uint64_t seq, uint8_t stage, uint64_t aux, uint64_t aux2) +{ + /* Routed through the same floor as the back-dated stamps, so the floor + * reflects every record in the ring and not just the ones we adjust. */ + stamp_at(seq, stage, aux, aux2, latrec_tnow()); +} + +/* Stamps whose time had to be clamped to keep the ring ascending. Non-zero + * means A1 is understated for that many slots. */ +inline uint64_t clamped() { return g_clamped.load(std::memory_order_relaxed); } + +#else /* !LIBE3_ENABLE_LATREC */ + +inline void stamp_at(uint64_t, uint8_t, uint64_t, uint64_t, uint64_t) {} +inline void stamp_now(uint64_t, uint8_t, uint64_t, uint64_t) {} +inline uint64_t clamped() { return 0; } + +#endif /* LIBE3_ENABLE_LATREC */ + +/* A1 entry: the slot's data existed and nothing had moved yet. + * + * aux carries the source-defined slot key, per the catalog. aux2 carries the + * codelet's entry stamp, which splits A1 into the part that is jbpf getting to + * the codelet and the part that is the codelet moving the data: + * + * jbpf dispatch RECORD_BEGIN.aux2 - RECORD_BEGIN t_ns + * convert + write PROCESS_BEGIN t_ns - RECORD_BEGIN.aux2 + */ +inline void record_begin(uint64_t seq, + uint32_t sfn, + uint16_t abs_slot, + uint64_t gnb_ts_ns, + uint64_t codelet_entry_ts_ns) +{ + stamp_at(seq, + LATREC_RECORD_BEGIN, + (static_cast(sfn) << 16) | abs_slot, + codelet_entry_ts_ns, + gnb_ts_ns); +} + +/* A1 exit / A2 entry: the slot's data was in place and the codelet was about to + * submit the descriptor. + * + * aux carries the dispatcher's poll stamp, splitting A2 into the jbpf ring + * transit and the queue wait: + * + * ring transit PROCESS_BEGIN.aux - PROCESS_BEGIN t_ns + * queue wait ENCODE_E3SM_BEGIN t_ns - PROCESS_BEGIN.aux + */ +inline void process_begin(uint64_t seq, uint64_t codelet_ts_ns, uint64_t dispatch_ts_ns, uint32_t bytes) +{ + stamp_at(seq, LATREC_PROCESS_BEGIN, dispatch_ts_ns, bytes, codelet_ts_ns); +} + +/* A2 exit / A3 entry. */ +inline void encode_begin(uint64_t seq, uint32_t payload_bytes) +{ + stamp_now(seq, + LATREC_ENCODE_E3SM_BEGIN, + payload_bytes, + static_cast(libe3::PduType::INDICATION_MESSAGE)); +} + +/* A3 exit. */ +inline void encode_done(uint64_t seq, std::size_t encoded_bytes) +{ + stamp_now(seq, + LATREC_ENCODE_E3SM_DONE, + static_cast(encoded_bytes), + static_cast(libe3::PduType::INDICATION_MESSAGE)); +} + +/* A3 failed: the slot produced no indication. Closes the row with a reason + * instead of leaving it looking like a lost record. */ +inline void encode_failed(uint64_t seq) +{ + stamp_now(seq, LATREC_SKIPPED, 0, LATREC_SKIP_ENCODE); +} + +/* Publish this slot's key to libe3, immediately before entering the library. + * + * This is the whole of the cross-component join: libe3 records whatever was + * last passed here into EMIT_ENTER's aux, which latrec2csv.py surfaces as the + * `origin_seq` column on the outbound leg. Without it the library's records + * are unattributable to the slot that produced them. + * + * It is a no-op on a thread with no ring open, which is the other reason + * open_ring() has to have run first. + */ +inline void bind_libe3(uint64_t seq) +{ + latrec_ctx_set(seq); +} + +/* The emit tail: subscriber fan-out bookkeeping after the payload has gone + * into the library. latrec2csv.py emits it as ENCODE_E3SM_DONE__WAIT_ENTER_us, + * because (ENCODE_E3SM_DONE -> WAIT_ENTER) is a declared extra hop on the + * source leg. + * + * Stamped once per slot, after the loop, so it covers every subscriber rather + * than just the first - aux carries how many there were, which is what makes + * the per-subscriber cost recoverable. + */ +inline void emit_tail(uint64_t seq, std::size_t subscribers) +{ + stamp_now(seq, LATREC_WAIT_ENTER, static_cast(subscribers), 0); +} + +} // namespace e3sm_l1kpm_trace + +#endif /* E3_SM_L1KPM_TRACE_H */ diff --git a/src/e3sm/slot_iq_pipeline.cpp b/src/e3sm/slot_iq_pipeline.cpp index b9b2eca..947a608 100644 --- a/src/e3sm/slot_iq_pipeline.cpp +++ b/src/e3sm/slot_iq_pipeline.cpp @@ -7,6 +7,8 @@ #include "slot_iq_pipeline.h" +#include "e3sm/l1_kpm/l1_kpm_trace.h" + #include #include #include @@ -66,11 +68,24 @@ bool SlotIqPipeline::load_codelets() { desc.priority = 1; desc.runtime_threshold = 0; - desc.num_in_io_channel = 0; /* no control input channel - codelet - * has no runtime config map; every UL - * slot fires a complete output. */ desc.num_linked_maps = 0; + /* Control-input channels. The codelet declares both unconditionally, so both + * must be present in the load request or the load fails -- even in + * Controller mode, where we simply never send on them. */ + desc.num_in_io_channel = 2; + std::strncpy(desc.in_io_channel[0].name, "shm_in", + sizeof(desc.in_io_channel[0].name) - 1); + std::memcpy(&desc.in_io_channel[0].stream_id, &shm_cfg_stream_id_, + sizeof(jbpf_io_stream_id_t)); + desc.in_io_channel[0].has_serde = false; + + std::strncpy(desc.in_io_channel[1].name, "sel_in", + sizeof(desc.in_io_channel[1].name) - 1); + std::memcpy(&desc.in_io_channel[1].stream_id, &slot_sel_stream_id_, + sizeof(jbpf_io_stream_id_t)); + desc.in_io_channel[1].has_serde = false; + desc.num_out_io_channel = 1; std::strncpy(desc.out_io_channel[0].name, "output_map", sizeof(desc.out_io_channel[0].name) - 1); @@ -186,14 +201,81 @@ void SlotIqPipeline::stop() { } } +void SlotIqPipeline::maybe_send_shm_cfg() { + if (shm_cfg_sent_ || config_.writer != e3config::ShmWriter::Gnb) { + return; + } + + struct e3_shm_cfg cfg = {}; + cfg.version = E3_SHM_CFG_VERSION; + cfg.epoch = config_.shm_epoch; + cfg.nof_symbols = config_.radio.nof_symbols; + cfg.nof_subcarriers = static_cast(config_.radio.nof_subcarriers()); + cfg.scale = config_.cbf16_scale; + std::snprintf(cfg.name, sizeof(cfg.name), "%s", config_.shm_name.c_str()); + + int rc = jbpf_io_channel_send_msg(io_ctx_, &shm_cfg_stream_id_, &cfg, sizeof(cfg)); + if (rc == 0) { + std::printf("[SlotIqPipeline] Sent e3_shm_cfg: name=%s epoch=%u %u sym x %u subc scale=%g\n", + cfg.name, cfg.epoch, cfg.nof_symbols, cfg.nof_subcarriers, + static_cast(cfg.scale)); + + /* Also send a DEFAULT selector. + * + * The codelet keeps the selector in a jbpf array map, which is + * zero-initialised and only written when a sel_in message arrives. The + * helper version-checks it (sel->version != E3_SLOT_SEL_VERSION -> refuse), + * so with nothing ever sent the zeroed struct makes the helper reject every + * slot: rows never get written, the codelet flags TRUNCATED, and no + * Indication is published. "No selector configured" must mean "publish + * everything", not "publish nothing". + * + * slot_mask = 0 and sfn_mod = 0 are exactly that default. Workstream F + * will replace this with the union of the subscribers' slot sets. */ + struct e3_slot_sel sel = {}; + sel.version = E3_SLOT_SEL_VERSION; + sel.nof_slots_per_frame = static_cast(config_.radio.slots_per_frame()); + sel.slot_mask = 0; /* publish every slot */ + sel.sfn_mod = 0; /* every frame */ + sel.sfn_offset = 0; + int src = jbpf_io_channel_send_msg(io_ctx_, &slot_sel_stream_id_, &sel, sizeof(sel)); + if (src == 0) { + std::printf("[SlotIqPipeline] Sent default e3_slot_sel: publish all slots " + "(%u slots/frame)\n", sel.nof_slots_per_frame); + } else { + std::fprintf(stderr, + "[SlotIqPipeline] Failed to send default e3_slot_sel (rc=%d) — the gNB " + "helper will refuse every slot until it arrives.\n", src); + return; /* retry both next batch; shm_cfg_sent_ stays false */ + } + + shm_cfg_sent_ = true; + } else { + std::fprintf(stderr, + "[SlotIqPipeline] Failed to send e3_shm_cfg (rc=%d), will retry. Until it " + "lands the gNB helper is not attached and publishes nothing.\n", rc); + } +} + void SlotIqPipeline::process_buffers(struct jbpf_io_stream_id* /*stream_id*/, void** bufs, int num_bufs) { - /* CLOCK_REALTIME ns at poll entry. Same clock domain as the - * codelet's timestamps so we can subtract directly. One read per - * batch - the dispatcher polls in tight bursts and per-buf clock - * reads aren't worth the syscall cost. */ + /* Must happen on this thread: jbpf_io_channel_send_msg requires the io_ctx + * owner's thread context, so it cannot be done from start(). Same + * lazy-on-first-buffer pattern as IqPipeline's PRB filter config. */ + maybe_send_shm_cfg(); + + /* CLOCK_MONOTONIC ns at poll entry. Same clock domain as the codelet's + * timestamps so we can subtract directly. One read per batch - the + * dispatcher polls in tight bursts and per-buf clock reads aren't worth the + * syscall cost. + * + * Was CLOCK_REALTIME. The whole RAN-side chain moved to CLOCK_MONOTONIC + * together (gNB gnb_ts_ns, jbpf_time_get_ns for codelet_ts_ns, this stamp, + * E3SMLayer1's handler entry, and the dApp's arrival age). Monotonic is what + * latrec stamps, so the stage CSV and the latrec rings share one clock, and + * it cannot step under NTP the way REALTIME can. */ struct timespec ts; - clock_gettime(CLOCK_REALTIME, &ts); + clock_gettime(CLOCK_MONOTONIC, &ts); const uint64_t dispatch_ts_ns = static_cast(ts.tv_sec) * 1000000000ULL + static_cast(ts.tv_nsec); @@ -201,7 +283,11 @@ void SlotIqPipeline::process_buffers(struct jbpf_io_stream_id* /*stream_id*/, (dispatch_ts_ns / 1000ULL) & 0x7FFFFFFFULL); for (int i = 0; i < num_bufs; i++) { - auto* sample = static_cast(bufs[i]); + /* The codelet's output element is `struct e3_slot_desc` — a ~64 B + * descriptor, not the old 733 KB uplink_slot_sample. The IQ never + * crosses this channel any more: the gNB-side helper converted the slot + * straight into /e3_ran_buffers and told us which row. */ + auto* sample = static_cast(bufs[i]); /* SPSC enqueue. Single producer here (this thread), single * consumer (the worker). Head is relaxed; tail load uses @@ -240,6 +326,14 @@ void SlotIqPipeline::process_buffers(struct jbpf_io_stream_id* /*stream_id*/, } void SlotIqPipeline::worker_loop() { + /* Open this thread's stage-record ring before the loop, not inside it: the + * call faults in the whole mapping, and it has to happen before the first + * indication is emitted or libe3 would open the ring first under its own + * name and every record we stamp would be attributed to the library + * instead of to this component. Idempotent, and a no-op in a build without + * the recorder compiled in. */ + e3sm_l1kpm_trace::open_ring(); + while (running_.load(std::memory_order_acquire)) { const uint64_t tail = queue_tail_.load(std::memory_order_relaxed); const uint64_t head = queue_head_.load(std::memory_order_acquire); @@ -264,7 +358,7 @@ void SlotIqPipeline::worker_loop() { void SlotIqPipeline::dispatch_sample(const QueueEntry& entry) { /* entry.buf points straight at the jbpf ring buffer (Option A, zero-copy). * Valid until the worker releases it right after this fan-out returns. */ - const auto& s = *static_cast(entry.buf); + const auto& s = *static_cast(entry.buf); SlotSample sample; sample.sfn = s.sfn; @@ -274,9 +368,19 @@ void SlotIqPipeline::dispatch_sample(const QueueEntry& entry) { sample.nof_ports = s.nof_ports; sample.nof_symbols = s.nof_symbols; sample.nof_subcarriers = s.nof_subcarriers; - sample.iq = s.iq; - sample.iq_size_bytes = s.iq_size_bytes; + + /* No IQ on this path. The gNB-side helper already wrote the fp16 row into + * /e3_ran_buffers; the descriptor says where. E3SMLayer1 forwards these + * indices to the dApp instead of converting anything itself. */ + sample.iq = nullptr; + sample.iq_size_bytes = 0; + sample.fh_buffer_index = s.fh_buffer_index; + sample.fh_write_index = s.fh_write_index; + sample.bytes_written = s.bytes_written; + sample.flags = s.flags; + sample.gnb_ts_ns = s.gnb_ts_ns; + sample.codelet_entry_ts_ns = s.codelet_entry_ts_ns; sample.codelet_ts_ns = s.codelet_ts_ns; sample.dispatch_ts_ns = entry.dispatch_ts_ns; sample.recv_us = entry.recv_us; diff --git a/src/e3sm/slot_iq_pipeline.h b/src/e3sm/slot_iq_pipeline.h index 3f25170..f535b09 100644 --- a/src/e3sm/slot_iq_pipeline.h +++ b/src/e3sm/slot_iq_pipeline.h @@ -25,6 +25,8 @@ #pragma once #include "../../codelets/uplink_slot_samples/uplink_slot_data.h" +#include "../../codelets/include/jbpf_e3_slot_api.h" +#include "e3_config.h" #include "jbpf_dispatcher.h" #include @@ -67,24 +69,53 @@ struct SlotSample { uint16_t nof_subcarriers{0}; /* Pointer into the worker's backing buffer. Valid for the callback - * duration only. cbf16_t = 4 bytes per complex sample. */ + * duration only. cbf16_t = 4 bytes per complex sample. + * + * NULL in `writer: gnb` mode: there the gNB-side helper has already + * converted the slot straight into /e3_ran_buffers, so no IQ crosses the + * jbpf ring and the descriptor below says where the row landed. */ const uint8_t* iq{nullptr}; uint32_t iq_size_bytes{0}; - /* RAN anchor (CLOCK_REALTIME ns) stamped by the ocudu hook caller - * at hand-off. Same domain as jbpf_time_get_ns() and the dApp's - * time.time_ns(), so it subtracts cleanly on both sides for true - * RAN -> dApp end-to-end latency. */ + /* Row the gNB-side helper wrote, in `writer: gnb` mode -- the + * (fh_buffer_index, fh_write_index) pair the Indication carries. + * + * Meaningless in `writer: controller` mode, where the controller writes the + * row itself and derives these from its own ring cursor. Exactly one writer + * may own that cursor, which is what e3config::ShmWriter enforces. */ + uint8_t fh_buffer_index{0}; + uint32_t fh_write_index{0}; + + /* Bytes the gNB helper actually wrote into the row, and the codelet's + * status flags. bytes_written == 0 with E3_SLOT_FLAG_TRUNCATED set means the + * helper REFUSED (geometry mismatch, unattached, or a slot larger than the + * row) rather than writing a partial one — publishing a prefix would be a + * silently wrong measurement. */ + uint32_t bytes_written{0}; + uint8_t flags{0}; + + /* A1 entry: RAN anchor stamped by the ocudu hook caller at hand-off, on the + * last symbol with the grid complete and nothing yet copied. + * + * CLOCK_MONOTONIC ns, same domain as jbpf_time_get_ns() (patched) and + * dispatch_ts_ns below, so the stage subtractions are exact. Note this is + * no longer comparable against a dApp's CLOCK_REALTIME clock without going + * through a ring header's mono/real pair. */ uint64_t gnb_ts_ns{0}; - /* Codelet-entry timestamp (CLOCK_REALTIME ns). Subtract gnb_ts_ns - * to get the jbpf invocation overhead. Used by the SM to log the - * gnb_to_codelet stage in statistics_layer1.log. */ + /* Codelet ENTRY timestamp (CLOCK_MONOTONIC ns). Subtract gnb_ts_ns for the + * jbpf invocation overhead alone -- this is what gnb_to_codelet_us reports. */ + uint64_t codelet_entry_ts_ns{0}; + + /* Codelet EXIT timestamp. (codelet_ts_ns - codelet_entry_ts_ns) is the + * codelet's own execution; in `writer: gnb` that is the convert + row write, + * reported as codelet_publish_us. Also the base for codelet_to_dispatch. */ uint64_t codelet_ts_ns{0}; /* Controller-side timestamps captured when the slot arrived at - * the dispatcher poll. dispatch_ts_ns = CLOCK_REALTIME ns at - * dispatcher entry (for stage timing vs. codelet_ts_ns). + * the dispatcher poll. dispatch_ts_ns = CLOCK_MONOTONIC ns at + * dispatcher entry (for stage timing vs. codelet_ts_ns). One read per poll + * batch, shared by every buffer in the batch. * recv_us = wall-clock microseconds at dispatcher entry (legacy, * kept for backwards-compatible stats consumers). * sample_id = monotonic count since startup. */ @@ -111,6 +142,24 @@ class SlotIqPipeline { /* CPU core to pin the worker thread on. -1 = no pinning. */ int worker_core{-1}; + + /* Who converts and writes the fp16 rows. In Gnb mode the pipeline sends + * the codelet an e3_shm_cfg so the gNB-side helper can attach to + * /e3_ran_buffers and write rows itself; in Controller mode it sends + * nothing and E3SMLayer1 does the conversion. */ + e3config::ShmWriter writer{e3config::ShmWriter::Controller}; + + /* Region identity + geometry handed to the gNB-side helper. Only used in + * Gnb mode. The controller OWNS the region, so it is the one that gets to + * say where it is and what shape it has; the helper cross-checks these + * against SharedMemoryHeader on attach and refuses a mismatch. */ + std::string shm_name{"/e3_ran_buffers"}; + e3config::RadioGeometry radio{}; + float cbf16_scale{1.0f}; + + /* Bumped by the controller whenever it recreates the region, so a helper + * holding a stale mapping refuses rather than writing into it. */ + uint16_t shm_epoch{1}; }; SlotIqPipeline(JbpfDispatcher& dispatcher, jbpf_io_ctx* io_ctx, Config cfg); @@ -156,7 +205,7 @@ class SlotIqPipeline { * this buffer; the worker releases it (jbpf_io_channel_release_buf) after * the consumer fan-out. Valid from enqueue until that release. */ void* buf; - /* CLOCK_REALTIME ns at dispatcher poll entry. Used by the SM to + /* CLOCK_MONOTONIC ns at dispatcher poll entry. Used by the SM to * derive the codelet -> dispatcher stage cost. */ uint64_t dispatch_ts_ns; uint32_t recv_us; @@ -176,6 +225,15 @@ class SlotIqPipeline { * (the worker won't see it); the worker releases the rest after fan-out. */ void process_buffers(struct jbpf_io_stream_id* sid, void** bufs, int n); + /* Send the e3_shm_cfg control-input message so the gNB-side helper attaches + * to /e3_ran_buffers. Gnb mode only; no-op otherwise. + * + * Called from process_buffers, i.e. on the dispatcher poll thread, NOT from + * start(): jbpf_io_channel_send_msg must run on the thread that owns the + * io_ctx. Same constraint and the same lazy-on-first-buffer pattern as + * IqPipeline's PRB filter config. Retried until it succeeds. */ + void maybe_send_shm_cfg(); + /* Worker-thread loop: drains the SPSC queue, fans out to each * registered consumer via the SlotSample callback. */ void worker_loop(); @@ -203,6 +261,25 @@ class SlotIqPipeline { /* Stream ID matches the codelet binary's compile-time UUID - * keep in sync with codelets/uplink_slot_samples/uplink_slot_samples.yaml * (stream_id: "7531abcd1234567890fedcba0987654a"). */ + /* Control-input channels the codelet declares (workstream B/C): + * shm_in -- struct e3_shm_cfg, tells the gNB helper where/what shape + * sel_in -- struct e3_slot_sel, temporal selector (workstream F) + * + * Distinct stream IDs from the output channel. Keep in sync with the + * jbpf_control_input_map names in uplink_slot_collect.c and with + * codelets/uplink_slot_samples/uplink_slot_samples.yaml. */ + struct jbpf_io_stream_id shm_cfg_stream_id_ { + 0x75, 0x31, 0xab, 0xcd, 0x12, 0x34, 0x56, 0x78, + 0x90, 0xfe, 0xdc, 0xba, 0x09, 0x87, 0x65, 0x4b + }; + struct jbpf_io_stream_id slot_sel_stream_id_ { + 0x75, 0x31, 0xab, 0xcd, 0x12, 0x34, 0x56, 0x78, + 0x90, 0xfe, 0xdc, 0xba, 0x09, 0x87, 0x65, 0x4c + }; + + /* e3_shm_cfg delivered? Retried on every buffer batch until it lands. */ + bool shm_cfg_sent_{false}; + struct jbpf_io_stream_id slot_iq_stream_id_ { .id = {0x75, 0x31, 0xab, 0xcd, 0x12, 0x34, 0x56, 0x78, 0x90, 0xfe, 0xdc, 0xba, 0x09, 0x87, 0x65, 0x4a} diff --git a/src/e3sm/sm_spectrum/e3sm_spect_wrapper.cpp b/src/e3sm/sm_spectrum/e3sm_spect_wrapper.cpp index c6c9434..3478429 100644 --- a/src/e3sm/sm_spectrum/e3sm_spect_wrapper.cpp +++ b/src/e3sm/sm_spectrum/e3sm_spect_wrapper.cpp @@ -11,7 +11,6 @@ extern "C" { #include "Spectrum-IQDataIndication.h" #include "Spectrum-PRBBlacklistControl.h" -#include "Spectrum-ConfigControl.h" #include "Spectrum-RanFunctionData.h" #include "aper_encoder.h" #include "aper_decoder.h" @@ -112,48 +111,6 @@ bool encode_spectrum_prb_blacklist_control(const SpectrumPRBBlacklistControl& in return ok; } -bool encode_spectrum_config_control(const SpectrumConfigControl& in, std::vector& out) { - Spectrum_ConfigControl_t cfg; - memset(&cfg, 0, sizeof(cfg)); - - // Set noiseFloorThreshold - cfg.noiseFloorThreshold = (long *)malloc(sizeof(long)); - if (!cfg.noiseFloorThreshold) return false; - *cfg.noiseFloorThreshold = in.noise_floor_threshold; - - // Set averagingFrames - cfg.averagingFrames = (long *)malloc(sizeof(long)); - if (!cfg.averagingFrames) { - free(cfg.noiseFloorThreshold); - return false; - } - *cfg.averagingFrames = in.averaging_frames; - - // Set enable - cfg.enable = (BOOLEAN_t *)malloc(sizeof(BOOLEAN_t)); - if (!cfg.enable) { - free(cfg.noiseFloorThreshold); - free(cfg.averagingFrames); - return false; - } - *cfg.enable = in.enable ? 1 : 0; - - uint8_t buffer[256]; - asn_enc_rval_t ret = aper_encode_to_buffer( - &asn_DEF_Spectrum_ConfigControl, NULL, &cfg, buffer, sizeof(buffer)); - - bool ok = (ret.encoded != -1); - if (ok) { - size_t bytes = (ret.encoded + 7) / 8; - out.assign(buffer, buffer + bytes); - } - - free(cfg.noiseFloorThreshold); - free(cfg.averagingFrames); - free(cfg.enable); - return ok; -} - bool encode_spectrum_ran_function_data(std::vector& out) { Spectrum_RanFunctionData_t rfd; diff --git a/src/e3sm/sm_spectrum/e3sm_spect_wrapper.h b/src/e3sm/sm_spectrum/e3sm_spect_wrapper.h index 691138c..f81cff1 100644 --- a/src/e3sm/sm_spectrum/e3sm_spect_wrapper.h +++ b/src/e3sm/sm_spectrum/e3sm_spect_wrapper.h @@ -27,12 +27,6 @@ struct SpectrumPRBBlacklistControl { int validity_period; // Validity in seconds (1..3600) }; -struct SpectrumConfigControl { - int noise_floor_threshold; // Noise floor threshold (-100..100) - int averaging_frames; // Averaging window (1..255) - bool enable; // Enable/disable monitoring -}; - struct SpectrumRanFunctionData { std::vector name; int version; @@ -45,9 +39,6 @@ bool encode_spectrum_iq_indication(const SpectrumIQIndication& in, std::vector& out); -// Encode Spectrum-ConfigControl into APER bytes -bool encode_spectrum_config_control(const SpectrumConfigControl& in, std::vector& out); - // Encode RAN function data into APER bytes bool encode_spectrum_ran_function_data(std::vector& out); diff --git a/src/e3sm/sm_spectrum/e3sm_spectrum.cpp b/src/e3sm/sm_spectrum/e3sm_spectrum.cpp index a68e691..75ceeba 100644 --- a/src/e3sm/sm_spectrum/e3sm_spectrum.cpp +++ b/src/e3sm/sm_spectrum/e3sm_spectrum.cpp @@ -6,8 +6,10 @@ #include "e3sm_spectrum.h" +#include #include #include +#include libe3::ErrorCode E3SMSpectrum::init() { // Register the pipeline consumer once at SM registration. IqPipeline keeps @@ -72,6 +74,20 @@ void E3SMSpectrum::on_sample(const e3sm_pipeline::DecompressedSample& s) { const auto subs = get_subscribers(); if (subs.empty()) return; // No one is listening. + using clock = std::chrono::steady_clock; + + // CLOCK_REALTIME ns at on_sample entry. Same domain as + // s.gnb_ts_ns / s.codelet_ts_ns / s.dispatch_ts_ns, so the three + // RAN-side stage durations subtract cleanly into statistics_spectrum.log. + // Mirrors E3SMLayer1's handler-entry stamp. + auto realtime_ns_now = []() -> uint64_t { + struct timespec ts; + clock_gettime(CLOCK_REALTIME, &ts); + return static_cast(ts.tv_sec) * 1000000000ULL + + static_cast(ts.tv_nsec); + }; + const uint64_t handler_entry_ns = realtime_ns_now(); + // RF=1 in-band IQ telemetry is APER-only (Spectrum-IQDataIndication). // There is no in-band IQ JSON encoder, so // warn once and drop if the agent is running JSON. @@ -121,12 +137,17 @@ void E3SMSpectrum::on_sample(const e3sm_pipeline::DecompressedSample& s) { indication_.timestamp = static_cast(s.raw->timestamp); encoded_buf_.clear(); - if (!e3sm_spectrum::encode_spectrum_iq_indication(indication_, encoded_buf_)) { + const auto t_encode_start = clock::now(); + const bool encoded_ok = + e3sm_spectrum::encode_spectrum_iq_indication(indication_, encoded_buf_); + const auto t_encode_end = clock::now(); + if (!encoded_ok) { std::fprintf(stderr, "[E3SMSpectrum] Failed to APER-encode IQ indication\n"); return; } // Fan out one indication per subscriber. + const auto t_emit_start = clock::now(); for (uint32_t dapp_id : subs) { libe3::Pdu pdu = make_indication_pdu(dapp_id, RAN_FUNCTION_ID, encoded_buf_); auto rc = emit_outbound(std::move(pdu)); @@ -136,6 +157,46 @@ void E3SMSpectrum::on_sample(const e3sm_pipeline::DecompressedSample& s) { dapp_id, libe3::error_code_to_string(rc)); } } + const auto t_emit_end = clock::now(); + + // --- Per-sample statistics (disabled unless --stats-log was given) --- + ++sample_publish_seq_; + if (stats_log_path_.empty()) { + return; + } + // Schema (matches E3SMLayer1's statistics_layer1.log columns for the + // three RAN-side stages; the trailing per-SM stage names differ): + // sample_seq, + // gnb_to_codelet_us, ocudu hook -> codelet entry (jbpf invocation) + // codelet_to_dispatch_us,codelet -> IqPipeline dispatcher poll + // dispatch_to_handler_us,dispatcher -> on_sample (SPSC queue wait) + // decompress_ns, BFP-9 decompression (IqPipeline worker) + // encode_ns, APER encode cost + // emit_ns, emit_outbound fan-out (SM-side enqueue) + // fft_size, zero-padded FFT bin count + // payload_bytes codelet-side compressed section size + if (!stats_log_.is_open()) { + stats_log_.open(stats_log_path_, std::ios::out | std::ios::trunc); + stats_log_ << "sample_seq," + "gnb_to_codelet_us,codelet_to_dispatch_us,dispatch_to_handler_us," + "decompress_ns,encode_ns,emit_ns,fft_size,payload_bytes\n"; + } + auto ns_between = [](auto a, auto b) { + return std::chrono::duration_cast(b - a).count(); + }; + auto sat_us = [](uint64_t lhs, uint64_t rhs) -> uint64_t { + return (lhs > rhs) ? ((lhs - rhs) / 1000ULL) : 0ULL; + }; + stats_log_ << sample_publish_seq_ << ',' + << sat_us(s.codelet_ts_ns, s.gnb_ts_ns) << ',' + << sat_us(s.dispatch_ts_ns, s.codelet_ts_ns) << ',' + << sat_us(handler_entry_ns, s.dispatch_ts_ns) << ',' + << s.decompress_ns << ',' + << ns_between(t_encode_start, t_encode_end) << ',' + << ns_between(t_emit_start, t_emit_end) << ',' + << fft_size << ',' + << s.raw->payload_size << '\n'; + stats_log_.flush(); } libe3::ErrorCode E3SMSpectrum::handle_control_action( diff --git a/src/e3sm/sm_spectrum/e3sm_spectrum.h b/src/e3sm/sm_spectrum/e3sm_spectrum.h index 31fc0b5..eea0db5 100644 --- a/src/e3sm/sm_spectrum/e3sm_spectrum.h +++ b/src/e3sm/sm_spectrum/e3sm_spectrum.h @@ -27,6 +27,7 @@ #include #include +#include #include #include @@ -34,8 +35,12 @@ class E3SMSpectrum : public libe3::ServiceModel { public: static constexpr uint32_t RAN_FUNCTION_ID = 1; - E3SMSpectrum(e3sm_pipeline::IqPipeline& pipeline, libe3::E3Agent& agent) - : pipeline_(pipeline), agent_(&agent) {} + E3SMSpectrum(e3sm_pipeline::IqPipeline& pipeline, + libe3::E3Agent& agent, + std::string stats_log_path = "") + : pipeline_(pipeline), + agent_(&agent), + stats_log_path_(std::move(stats_log_path)) {} std::string name() const override { return "Spectrum Service Model"; } uint32_t version() const override { return 1; } @@ -70,6 +75,14 @@ class E3SMSpectrum : public libe3::ServiceModel { std::vector padded_buf_; std::vector encoded_buf_; e3sm_spectrum::SpectrumIQIndication indication_; + + // Per-slot RAN-side stage CSV. Mirrors E3SMLayer1's statistics_layer1.log + // schema (see on_sample() in the .cpp for column order). Disabled unless + // constructed with a non-empty path; opened lazily on first published + // sample so we don't create an empty file on runs that never receive one. + std::string stats_log_path_; + std::ofstream stats_log_; + uint64_t sample_publish_seq_{0}; }; #endif /* E3_SM_SPECTRUM_H */ diff --git a/src/e3sm/utils/e3sm_shm_writer.cpp b/src/e3sm/utils/e3sm_shm_writer.cpp index ec8cadb..1faeb90 100644 --- a/src/e3sm/utils/e3sm_shm_writer.cpp +++ b/src/e3sm/utils/e3sm_shm_writer.cpp @@ -119,38 +119,21 @@ ShmIqWriter::~ShmIqWriter() { close(); } -bool ShmIqWriter::open(const std::string& shm_name, size_t total_size) { +bool ShmIqWriter::open(const std::string& shm_name, size_t total_size, + const e3config::RadioGeometry& geom, float cbf16_scale, + e3config::ShmWriter writer) { if (mapped_ != nullptr) { std::fprintf(stderr, "[ShmIqWriter] open() called twice\n"); return false; } shm_name_ = shm_name; - // Pick up the cbf16→fp16 scale factor from the env. We parse here - // rather than in publish_row_cbf16 so the strtof cost (and the - // "no env var → default" message) happens once at startup, not - // per slot. Invalid / non-positive values fall back to 1.0 with - // a warning - 0 or negative would zero the published row and - // silently break the dApp. - if (const char* env = std::getenv("E3_CBF16_SCALE"); env != nullptr) { - char* end = nullptr; - float parsed = std::strtof(env, &end); - if (end != env && std::isfinite(parsed) && parsed > 0.0f) { - cbf16_scale_ = parsed; - std::printf("[ShmIqWriter] cbf16 → fp16 scale = %g (from E3_CBF16_SCALE)\n", - static_cast(cbf16_scale_)); - } else { - std::fprintf(stderr, - "[ShmIqWriter] WARNING: invalid E3_CBF16_SCALE=%s, " - "using default %g\n", - env, static_cast(cbf16_scale_)); - } - } else { - std::printf("[ShmIqWriter] cbf16 → fp16 scale = %g " - "(default; set E3_CBF16_SCALE to tune)\n", - static_cast(cbf16_scale_)); - } - + // The cbf16->fp16 scale and the row geometry now come from the config + // (assigned below), not from an E3_CBF16_SCALE env var. They are part of a + // cross-process contract: the same values are pushed to the gNB-side publish + // helper via e3_shm_cfg, and a disagreement would produce different rows + // with no error -- bf16 and fp16 are both 2 bytes, so a wrong scale is + // silently wrong data. The loader rejects a non-positive scale. fd_ = ::shm_open(shm_name.c_str(), O_CREAT | O_RDWR, 0666); if (fd_ < 0) { std::fprintf(stderr, "[ShmIqWriter] shm_open(%s) failed: %s\n", @@ -183,9 +166,11 @@ bool ShmIqWriter::open(const std::string& shm_name, size_t total_size) { header_ = static_cast(mapped_); // Layout sizing. - num_fh_samples_ = static_cast(kShmAntsLayout) * kShmSymbolsPerRow - * kShmScPerSymbol * 2; // 366912 fp16's per row - row_bytes_ = num_fh_samples_ * sizeof(uint16_t); + geom_ = geom; + cbf16_scale_ = cbf16_scale; + writer_ = writer; + num_fh_samples_ = geom_.num_fh_samples(); // whole-row uint16 count + row_bytes_ = geom_.row_bytes(); num_buffers_ = 2; // double-buffered ring size_t usable = (total_size > sizeof(SharedMemoryHeader)) @@ -216,6 +201,9 @@ bool ShmIqWriter::open(const std::string& shm_name, size_t total_size) { next_row_ = 0; next_buf_ = 0; + std::printf("[ShmIqWriter] writer=%s (%s)\n", e3config::to_string(writer_), + writes_rows() ? "this process converts and writes rows" + : "gNB helper writes rows; this process owns the region only"); std::printf("[ShmIqWriter] %s opened (%zu bytes): %u buffers × %u rows × " "%u bytes/row (num_fh_samples=%u)\n", shm_name.c_str(), total_size, @@ -241,38 +229,58 @@ void ShmIqWriter::close() { mapped_size_ = 0; } -void ShmIqWriter::publish_row(const int16_t* iq_int16, +bool ShmIqWriter::publish_row(const int16_t* iq_int16, uint8_t& out_buffer_index, uint32_t& out_write_index) { + if (!writes_rows()) { + return false; + } out_buffer_index = next_buf_; out_write_index = next_row_; - // Antenna-0 region of the target row. The remaining 3 antennas are kept - // at zero (set once in open()); the dApp only reads antenna 0. + // Antenna-0 region of the target row; the dApp only reads antenna 0 on this + // legacy path. uint8_t* row_base = buffers_base_ + static_cast(next_buf_) * fh_buffer_size_ + static_cast(next_row_) * row_bytes_; uint16_t* ant0 = reinterpret_cast(row_base); - const size_t n_pairs = static_cast(kShmSymbolsPerRow) * kShmScPerSymbol; + const size_t n_pairs = static_cast(geom_.nof_symbols) * geom_.nof_subcarriers(); for (size_t i = 0; i < n_pairs; ++i) { ant0[i * 2 + 0] = int16_to_fp16(iq_int16[i * 2 + 0]); ant0[i * 2 + 1] = int16_to_fp16(iq_int16[i * 2 + 1]); } + // Clear the antennas this path never writes. The previous comment claimed + // they "are kept at zero (set once in open())" — true only the first time a + // row is used. Rows recycle through the ring, so without this the tail holds + // IQ from an OLDER slot and the dApp cannot distinguish it from a quiet + // antenna. See the same fix in publish_row_cbf16 and in the gNB-side helper. + const size_t written = n_pairs * 2u * sizeof(uint16_t); + if (written < row_bytes_) { + std::memset(row_base + written, 0, row_bytes_ - written); + } + // Advance the ring. Buffer roll-over only when we wrap past the last row. if (++next_row_ >= num_fh_rows_) { next_row_ = 0; next_buf_ = static_cast((next_buf_ + 1) % num_buffers_); } + return true; } -void ShmIqWriter::publish_row_cbf16(const uint8_t* iq_cbf16_bytes, +bool ShmIqWriter::publish_row_cbf16(const uint8_t* iq_cbf16_bytes, uint16_t nof_ports, uint8_t& out_buffer_index, uint32_t& out_write_index) { + // Refuse in `writer: gnb` mode. The gNB-side helper owns the ring cursor + // there; advancing ours too would have both processes writing different rows + // and reporting indices the other invalidates -- with no error anywhere. + if (!writes_rows()) { + return false; + } out_buffer_index = next_buf_; out_write_index = next_row_; @@ -294,12 +302,12 @@ void ShmIqWriter::publish_row_cbf16(const uint8_t* iq_cbf16_bytes, // bf16 -> fp16 transformation). This is also the per-antenna fp16 stride // within the row (== kShmAntStride): both src ([port][..]) and dst // ([ant][..]) advance by n_u16 per antenna, so port p maps to antenna p. - const size_t n_u16 = static_cast(kShmSymbolsPerRow) * kShmScPerSymbol * 2u; + const size_t n_u16 = geom_.u16_per_ant(); // Write every delivered antenna; antennas >= nof_ports stay zero (the row // was zero-filled once in open()). Clamp to the row's antenna capacity. uint16_t ports = nof_ports ? nof_ports : 1; - if (ports > kShmAntsLayout) ports = kShmAntsLayout; + if (ports > geom_.nof_ports) ports = geom_.nof_ports; for (uint16_t a = 0; a < ports; ++a) { uint16_t* dst = row_u16 + static_cast(a) * n_u16; @@ -321,11 +329,15 @@ void ShmIqWriter::publish_row_cbf16(const uint8_t* iq_cbf16_bytes, __m128i ph = _mm256_cvtps_ph(fscaled, _MM_FROUND_TO_NEAREST_INT); _mm_storeu_si128(reinterpret_cast<__m128i*>(dst + i), ph); } - // n_u16 is a multiple of 8 per antenna (273 PRB -> 91728), so the AVX2 - // loop handles every element; assert it and skip a scalar tail. - static_assert((kShmSymbolsPerRow * kShmScPerSymbol * 2) % 8 == 0, - "n_u16 must be a multiple of 8 for the AVX2 path"); - (void)i; + // n_u16 must be a multiple of 8 for the AVX2 loop to cover every + // element with no scalar tail (273 PRB x 14 sym x 2 = 91,728 = 8 x + // 11,466). Now that the geometry is runtime, this is enforced up front + // by e3config::RadioGeometry::validate() rather than by static_assert, + // so a configuration that would silently take a scalar remainder is + // rejected at startup instead. + for (; i < n_u16; ++i) { + dst[i] = bf16_to_fp16(srca[i], scale); + } #else for (size_t i = 0; i < n_u16; ++i) { dst[i] = bf16_to_fp16(srca[i], scale); @@ -333,10 +345,19 @@ void ShmIqWriter::publish_row_cbf16(const uint8_t* iq_cbf16_bytes, #endif } + // Clear antenna slots this slot did not fill — see publish_row for why + // "zero-filled once in open()" is not sufficient once rows recycle. Free at + // full occupancy (4 ports into a 4-antenna row leaves no tail). + const size_t written = static_cast(ports) * n_u16 * sizeof(uint16_t); + if (written < row_bytes_) { + std::memset(row_base + written, 0, row_bytes_ - written); + } + if (++next_row_ >= num_fh_rows_) { next_row_ = 0; next_buf_ = static_cast((next_buf_ + 1) % num_buffers_); } + return true; } } // namespace e3sm_spectrum diff --git a/src/e3sm/utils/e3sm_shm_writer.h b/src/e3sm/utils/e3sm_shm_writer.h index 2095ac0..2999f64 100644 --- a/src/e3sm/utils/e3sm_shm_writer.h +++ b/src/e3sm/utils/e3sm_shm_writer.h @@ -20,20 +20,23 @@ #include #include +#include "e3_config.h" + namespace e3sm_spectrum { -// Layout constants — must match the dApp's compile-time constants in -// subcarrier_power_app.cpp (N_ANTS, N_SYMBOLS, N_PRBS, N_SC_PER_PRB). -// Hardcoded to the srsRAN-janus 100 MHz @ 30 kHz SCS config (273 PRBs). -// If the RAN bandwidth changes, both this constant AND the dApp's N_PRBS -// must be updated in lockstep so the SHM row stride matches. -constexpr int kShmAntsLayout = 4; // dApp's expected_u16 uses N_ANTS=4 -constexpr int kShmSymbolsPerRow = 14; -constexpr int kShmPrbsPerSymbol = 273; -constexpr int kShmScPerPrb = 12; -constexpr int kShmScPerSymbol = kShmPrbsPerSymbol * kShmScPerPrb; // 3276 -constexpr int kShmAntStride = kShmSymbolsPerRow * kShmScPerSymbol * 2; -constexpr int kShmSymStride = kShmScPerSymbol * 2; +// The row geometry used to live here as four `constexpr` values with a comment +// requiring lockstep updates with the dApp's N_PRBS. It is now runtime state, +// taken from the YAML config at open() (e3config::RadioGeometry) and written +// into SharedMemoryHeader for consumers to read back. +// +// Two reasons the constants had to go: +// - changing the antenna count or bandwidth required a recompile, and +// - they were already inconsistent with the rest of the controller: +// `--num-prbs` was accepted on the command line and fed only the eCPRI PRB +// filter, while this file kept sizing rows from `kShmPrbsPerSymbol = 273`. +// +// The config is a bootstrap, not the truth — E3SMLayer1 validates it against +// the geometry the RAN reports in each slot. See e3_config.h. // Mirrors SharedMemoryHeader in // spear-aerial-sample-apps/dapps/common/e3_manager/e3_manager.h:39. @@ -61,7 +64,23 @@ class ShmIqWriter { // shm_open + ftruncate + mmap + write header. Returns false on failure // (with errno set + a message on stderr). - bool open(const std::string& shm_name, size_t total_size); + // + // `geom` fixes the row stride and antenna capacity; `cbf16_scale` is the + // bf16 -> fp16 factor, previously read from the E3_CBF16_SCALE env var and + // now part of the config because the gNB-side publish helper needs the same + // value and a disagreement would be silently wrong data. + bool open(const std::string& shm_name, size_t total_size, + const e3config::RadioGeometry& geom, float cbf16_scale, + e3config::ShmWriter writer = e3config::ShmWriter::Controller); + + // True when this process is the one converting and writing rows. False in + // `writer: gnb` mode, where the controller owns the region but the gNB-side + // helper writes it. + // + // The publish_row* methods HARD-REFUSE when this is false. Both writers keep + // their own ring cursor, so two active writers would silently overwrite each + // other's rows -- making that impossible is the whole point of the switch. + bool writes_rows() const { return writer_ == e3config::ShmWriter::Controller; } void close(); @@ -69,7 +88,8 @@ class ShmIqWriter { // the next ring slot. iq_int16 is laid out as [sym][sc][I,Q] (3276*2*14 // int16 values). Returns the (fh_buffer_index, fh_write_index) that the // dApp will use to address the row. - void publish_row(const int16_t* iq_int16, + // Returns false (writing nothing) when writes_rows() is false. + bool publish_row(const int16_t* iq_int16, uint8_t& out_buffer_index, uint32_t& out_write_index); @@ -77,11 +97,15 @@ class ShmIqWriter { // codelet-produced cbf16_t blob (bf16 real + bf16 imag, 4 bytes per // complex sample) laid out as [port][sym][sc][I,Q] - matches ocudu's // resource_grid_reader_impl tensor layout. Writes nof_ports antennas - // (clamped to kShmAntsLayout); antennas beyond nof_ports stay zero. + // (clamped to the configured antenna capacity); antennas beyond nof_ports + // are zeroed on every publish -- NOT merely left alone. Rows recycle through + // the ring, so a slot with fewer ports than the row holds would otherwise + // expose IQ from an OLDER slot as if it were a quiet antenna. // Converts bf16 → IEEE half (fp16) on the fly so the dApp's reader // sees the same wire shape it expects from the legacy int16 path. // Returns the (fh_buffer_index, fh_write_index) like publish_row. - void publish_row_cbf16(const uint8_t* iq_cbf16_bytes, + // Returns false (writing nothing) when writes_rows() is false. + bool publish_row_cbf16(const uint8_t* iq_cbf16_bytes, uint16_t nof_ports, uint8_t& out_buffer_index, uint32_t& out_write_index); @@ -89,6 +113,11 @@ class ShmIqWriter { uint32_t num_fh_rows() const { return num_fh_rows_; } uint32_t num_buffers() const { return num_buffers_; } uint32_t num_fh_samples() const { return num_fh_samples_; } + uint32_t row_bytes() const { return row_bytes_; } + + // Geometry this writer was opened with. E3SMLayer1 uses it instead of the + // old compile-time constants when sizing its expectations. + const e3config::RadioGeometry& geometry() const { return geom_; } // Scale factor applied to bf16 values in publish_row_cbf16 before // they are converted to fp16. ocudu's resource grid stores cbf16 @@ -97,8 +126,8 @@ class ShmIqWriter { // scale (65504). Tuning this knob shifts the on-wire fp16 // magnitudes up/down by a constant linear factor so the dApp's // dBFS readouts land in the displayable [-110, -10] range. Tuned - // by env var E3_CBF16_SCALE (read at open() time). Default 1.0 - // means "pass through unchanged". + // Supplied by the config (shm.cbf16_scale). Default 1.0 means + // "pass through unchanged". float cbf16_scale() const { return cbf16_scale_; } private: @@ -120,9 +149,14 @@ class ShmIqWriter { uint32_t next_row_ = 0; uint8_t next_buf_ = 0; - // See cbf16_scale() above for semantics. Set in open() from - // E3_CBF16_SCALE env var (default 1.0). + // See cbf16_scale() above for semantics. Set in open() from the config. float cbf16_scale_ = 1.0f; + + // Row geometry (antenna count, symbols, subcarriers) from the config. + e3config::RadioGeometry geom_{}; + + // Who writes rows. See writes_rows(). + e3config::ShmWriter writer_ = e3config::ShmWriter::Controller; }; } // namespace e3sm_spectrum diff --git a/start_e3controller_example.sh b/start_e3controller_example.sh index b7346ff..1df5700 100644 --- a/start_e3controller_example.sh +++ b/start_e3controller_example.sh @@ -1,25 +1,35 @@ -cd "$(dirname "$0")" +#!/bin/sh +# Launch the E3Controller. +# +# All configuration now lives in a YAML file — `--config` is the only argument. +# The 20 CLI options this script used to pass are gone; see +# configs/e3_controller.yaml for the schema, and include/e3_config.h for why the +# radio geometry belongs in one place. +# +# usage: ./start_e3controller_example.sh [config.yaml] -ENCODING="${1:-json}" -case "$ENCODING" in - json) SETUP_PORT=5555; PUB_PORT=5556; SUB_PORT=5557 ;; - asn1) SETUP_PORT=9990; PUB_PORT=9991; SUB_PORT=9999 ;; - *) echo "usage: $0 [json|asn1]" >&2; exit 2 ;; -esac -echo "=== E3Controller: encoding=$ENCODING ports setup=$SETUP_PORT pub=$PUB_PORT sub=$SUB_PORT ===" +cd "$(dirname "$0")" || exit 1 -./e3_controller \ - --encoding "$ENCODING" \ - --link-layer zmq \ - --transport tcp \ - --num-prbs \ - --shm-name /e3_ran_buffers \ - --shm-size $((1<<30)) \ - --codelet-path /workspace/e3_release/E3Controller/codelets \ - --poll-core \ - --worker-core \ - --publisher-core \ - --setup-port "$SETUP_PORT" \ - --publisher-port "$PUB_PORT" \ - --subscriber-port "$SUB_PORT" \ - --lcm-socket /dev/shm/jbpf/jbpf_lcm_ipc \ \ No newline at end of file +CONFIG="${1:-configs/e3_controller.yaml}" + +if [ ! -f "$CONFIG" ]; then + echo "config not found: $CONFIG" >&2 + echo "usage: $0 [config.yaml]" >&2 + exit 2 +fi + +# Reminders for things this file no longer sets, because they are in the YAML: +# +# - encoding: `e3.encoding` (asn1 | json). Note the port convention differs — +# a JSON/cuBB dApp expects 5555/5556/5557, ASN.1 uses 9990/9991/9999. Both +# are in the `e3:` section, so keep one config file per encoding rather than +# trying to override them here. +# - radio geometry: `radio:` — must match the running gNB. The controller +# validates it against the first slot the RAN delivers and refuses on a +# mismatch, so a wrong value is a startup error rather than a wrong spectrum. +# - cbf16 scale: `shm.cbf16_scale`, formerly the E3_CBF16_SCALE env var. It is +# now pushed down to the gNB-side publish helper, so setting the old env var +# has no effect. + +echo "=== E3Controller: config=$CONFIG ===" +exec ./e3_controller --config "$CONFIG"