Rosenblatt: FIRST CORRECT INT8 MATMUL ON THE MAINLINE ROCKET NPU

The whole path works end-to-end on real silicon (boltzmann, 7.0.0-rc3 rocket,
/dev/accel/accel0): our golden regcmd builder + rkt_operands tiling + librocket
submission -> a 4x8x16 INT8 matmul whose per-element output (incl requant +
int8 saturation) matches the CPU reference exactly (144,160,...,255).

ROOT CAUSE of the long all-zero saga: rocket_prep_bo()'s timeout_ns is an
ABSOLUTE CLOCK_MONOTONIC deadline (kernel drm_timeout_abs_to_jiffies), not a
relative duration. hwtest passed 3e9 ('3s') -> read as a deadline ~9h in the
past -> effective timeout 0 -> we read the output BO (memset 0) BEFORE the NPU
wrote it. The -EBUSY we dismissed WAS the 'not done' signal. Mesa passes
INT64_MAX (block until the DPU-write fence signals). Nothing was wrong with the
regcmd, operands, addresses, driver, or hardware -- diagnosed via the working
Mesa Teflon reference ('if tflite works it has the proper spells', per Markus;
divergence found by Fable/rknpu-specialist).

Fixes: hwtest prep_bo INT64_MAX; close all BO handles (kills drm_mm_takedown
WARN); librocket rocket_close_bo added + prep_bo doc corrected (abs deadline).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Markus Fritsche
2026-07-14 08:36:14 +02:00
parent 2b34125760
commit 1fb810d93c
3 changed files with 46 additions and 39 deletions
+32 -38
View File
@@ -1,11 +1,16 @@
/* SPDX-License-Identifier: MIT
*
* hwtest.c — run a real INT8 matmul on the mainline rocket NPU and verify.
* hwtest.c — VERIFIED INT8 matmul on the mainline rocket NPU.
*
* Y[M][N] = X[M][K] * W[K][N], expressed as a 1x1 conv (input_height=M,
* input_channels=K, output_channels=N). Operands are laid out in the NPU's
* tiled format (rkt_operands.c), submitted via our golden regcmd, and the
* de-tiled output is compared against a CPU reference.
* Y[M][N] = X[M][K] * W[K][N] expressed as a 1x1 conv; operands laid out in
* the NPU tiled format (rkt_operands), regcmd from our golden builder,
* submitted via librocket to /dev/accel/accel0; output checked vs a CPU ref.
*
* KEY GOTCHA (cost a long debug): rocket_prep_bo()'s timeout_ns is an
* ABSOLUTE CLOCK_MONOTONIC deadline (kernel drm_timeout_abs_to_jiffies), NOT
* a relative duration. Pass INT64_MAX to block until the DPU-write fence
* signals. A small "relative-looking" value is a deadline in the past =>
* timeout 0 => you read the buffer before the NPU has written it.
*/
#define _GNU_SOURCE
#include <stdint.h>
@@ -22,14 +27,12 @@
#define DEV "/dev/accel/accel0"
/* dims: M rows, K contraction, N outputs; 1x1 conv (iw=ow=1) */
enum { M = 4, K = 16, N = 8, IW = 1, OW = 1, WW = 1, WH = 1 };
enum { IZP = 128, WZP = 128, OZP = 128 };
struct bo {
uint32_t handle;
uint64_t dma;
uint64_t off;
uint64_t dma, off;
void *map;
uint32_t size;
};
@@ -37,15 +40,14 @@ struct bo {
static int mkbo(int fd, struct bo *b, uint32_t size)
{
b->size = (size + 0xfff) & ~0xfffu;
int r = rocket_create_bo(fd, b->size, &b->handle, &b->dma, &b->off);
if (r)
return r;
if (rocket_create_bo(fd, b->size, &b->handle, &b->dma, &b->off))
return -1;
b->map = rocket_mmap_bo(fd, b->off, b->size);
return b->map == MAP_FAILED ? -1 : 0;
}
static void put(int fd, struct bo *b, const void *src, unsigned n)
{
rocket_prep_bo(fd, b->handle, -1);
rocket_prep_bo(fd, b->handle, INT64_MAX);
memset(b->map, 0, b->size);
if (src)
memcpy(b->map, src, n);
@@ -54,21 +56,18 @@ static void put(int fd, struct bo *b, const void *src, unsigned n)
int main(void)
{
/* ---- host operands ---- */
uint8_t X[IW * M * K]; /* [iw][ih=M][ic=K] */
uint8_t Wt[N * WW * WH * K]; /* [oc=N][ww][wh][ic=K] */
/* operands: X = signed +1 everywhere; W[n][k] = signed (n+1) */
uint8_t X[IW * M * K], Wt[N * WW * WH * K];
int32_t Bin[N];
for (int i = 0; i < IW * M * K; i++)
X[i] = 128;
for (int m = 0; m < M; m++)
X[m * K + 0] = 128 + 1; /* signed +1 at channel 0 */
memset(Wt, 128, sizeof Wt);
X[i] = IZP + 1;
for (int n = 0; n < N; n++)
Wt[n * K + 0] = 128 + (n + 1); /* signed +(n+1) at channel 0 */
for (int k = 0; k < K; k++)
Wt[n * K + k] = WZP + (n + 1);
for (int n = 0; n < N; n++)
Bin[n] = 0;
/* ---- CPU reference (identity requant; izp=128 => no bias correction) ---- */
/* CPU reference (izp=128 => no bias correction; identity requant) */
uint8_t ref[M * OW * N];
for (int m = 0; m < M; m++)
for (int n = 0; n < N; n++) {
@@ -82,7 +81,6 @@ int main(void)
ref[m * N + n] = (uint8_t)(acc + 128);
}
/* ---- pack into NPU layout ---- */
unsigned wsz = rkt_packed_weights_size(WW, WH, K, N);
unsigned isz = rkt_raw_input_size(IW, M, K);
unsigned osz = rkt_raw_output_size(OW, M, N);
@@ -92,35 +90,30 @@ int main(void)
rkt_pack_input(X, IW, M, K, IZP, ipk);
rkt_compute_biases(Wt, Bin, WW, WH, K, N, WZP, IZP, bpk);
/* ---- open + allocate (guard first so nothing sits at IOVA 0) ---- */
int fd = rocket_open(DEV);
if (fd < 0) { printf("open failed %d\n", fd); return 1; }
/* guard BO first so nothing lands at NPU IOVA 0 */
struct bo guard = {0}, in = {0}, wt = {0}, bias = {0}, out = {0}, reg = {0};
if (mkbo(fd, &guard, 0x1000) || mkbo(fd, &wt, wsz) || mkbo(fd, &bias, sizeof bpk) ||
mkbo(fd, &in, isz) || mkbo(fd, &out, osz) || mkbo(fd, &reg, 0x1000)) {
printf("bo alloc failed\n"); return 1;
}
printf("dmas: in=%#x wt=%#x bias=%#x out=%#x reg=%#x (guard=%#x)\n",
(uint32_t)in.dma, (uint32_t)wt.dma, (uint32_t)bias.dma,
(uint32_t)out.dma, (uint32_t)reg.dma, (uint32_t)guard.dma);
put(fd, &wt, wpk, wsz);
put(fd, &in, ipk, isz);
put(fd, &bias, bpk, sizeof bpk);
put(fd, &out, NULL, 0);
/* ---- golden regcmd with real addresses; patch bias addr ---- */
uint64_t rc[512];
int n = rkt_build_matmul_regcmd(rc, 512, M, N, K, in.dma, wt.dma, out.dma,
IZP, WZP, OZP);
if (n < 0) { printf("regcmd build failed\n"); return 1; }
/* builder hardcodes biases_addr=0; point at the real bias BO */
for (int i = 0; i < n; i++)
if ((uint32_t)(rc[i] & 0xffff) == REG_DPU_RDMA_RDMA_BS_BASE_ADDR)
rc[i] = (rc[i] & ~(0xffffffffULL << 16)) |
((uint64_t)(uint32_t)bias.dma << 16);
put(fd, &reg, rc, (unsigned)n * sizeof(uint64_t));
/* ---- submit ---- */
struct drm_rocket_task task = { .regcmd = (uint32_t)reg.dma,
.regcmd_count = (uint32_t)n };
uint32_t inh[] = { in.handle, wt.handle, bias.handle, reg.handle };
@@ -135,14 +128,11 @@ int main(void)
job.out_bo_handles = (uintptr_t)outh;
job.out_bo_handle_count = 1;
int sr = rocket_submit(fd, &job, 1);
printf("SUBMIT rc=%d\n", sr);
if (sr) return 1;
int wr = rocket_prep_bo(fd, out.handle, 3000000000LL);
printf("WAIT rc=%d %s\n", wr, wr == 0 ? "== COMPLETED ==" : "(timeout/err)");
if (wr) return 2;
if (rocket_submit(fd, &job, 1)) { printf("submit failed\n"); return 1; }
/* ABSOLUTE deadline: INT64_MAX = block until the DPU-write fence signals */
int wr = rocket_prep_bo(fd, out.handle, INT64_MAX);
if (wr) { printf("wait failed %d\n", wr); return 2; }
/* ---- de-tile + verify ---- */
uint8_t got[M * OW * N];
rkt_unpack_output(out.map, OW, M, N, got);
int bad = 0;
@@ -157,7 +147,11 @@ int main(void)
}
printf("\n");
}
printf(bad ? "\nMISMATCH: %d elems off by >1\n" : "\nMATCH (within +/-1) — NPU MATMUL CORRECT\n", bad);
rocket_fini_bo(fd, out.handle);
printf(bad ? "\n%d MISMATCH(es)\n" : "\nNPU MATMUL CORRECT (matches CPU ref within +/-1)\n", bad);
/* release all BO handles (avoids drm_mm_takedown WARN at close) */
uint32_t all[] = { guard.handle, in.handle, wt.handle, bias.handle, out.handle, reg.handle };
for (unsigned i = 0; i < sizeof all / sizeof all[0]; i++)
rocket_close_bo(fd, all[i]);
return bad ? 3 : 0;
}
+10
View File
@@ -94,3 +94,13 @@ int rocket_submit(int fd, struct drm_rocket_job *jobs, uint32_t job_count)
return -errno;
return 0;
}
int rocket_close_bo(int fd, uint32_t handle)
{
struct drm_gem_close c;
memset(&c, 0, sizeof c);
c.handle = handle;
if (ioctl(fd, DRM_IOCTL_GEM_CLOSE, &c) < 0)
return -errno;
return 0;
}
+4 -1
View File
@@ -52,7 +52,7 @@ void *rocket_mmap_bo(int fd, uint64_t mmap_offset, uint32_t size);
* rocket_prep_bo - take CPU ownership of a BO (wait for NPU, sync caches)
* @fd: rocket device fd
* @handle: GEM handle of the BO
* @timeout_ns: max wait time in nanoseconds (-1 for infinite)
* @timeout_ns: ABSOLUTE CLOCK_MONOTONIC deadline (ns); INT64_MAX = block until signalled
*
* Returns 0 on success, -errno on failure.
*/
@@ -77,6 +77,9 @@ int rocket_fini_bo(int fd, uint32_t handle);
*/
int rocket_submit(int fd, struct drm_rocket_job *jobs, uint32_t job_count);
/* Release a BO handle (DRM_IOCTL_GEM_CLOSE). Returns 0 or -errno. */
int rocket_close_bo(int fd, uint32_t handle);
#ifdef __cplusplus
}
#endif