From 3ecaf4061ab39c0f262d7d5928d8049e9f8fd6cd Mon Sep 17 00:00:00 2001 From: Dmitry Ilyin <6576495+widgetii@users.noreply.github.com> Date: Mon, 13 Apr 2026 11:39:23 +0300 Subject: [PATCH] openrknn: document multi-core dispatch requirement for FP16 models Single-core NPU tasks only cover ~512 of 1024 spatial positions for SmolVLM transformer shards. The vendor runs all 3 NPU cores via kernel-level multi-core dispatch configured during rknn_init. openrknn's OWN mode skips vendor init, so core_mask=0 defaults to single core. Add TODO comment tracking this as the final blocker for correct FP16 output. The DMA address patching is now byte-exact (0 diffs) and the per- channel computation is verified correct (768/768 channel bias match). Co-Authored-By: Claude Opus 4.6 (1M context) --- openrknn/src/openrknn_run.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/openrknn/src/openrknn_run.c b/openrknn/src/openrknn_run.c index 9c37feb..e572313 100644 --- a/openrknn/src/openrknn_run.c +++ b/openrknn/src/openrknn_run.c @@ -2082,6 +2082,13 @@ int orknn_own_run(struct orknn_context *ctx, rknn_run_extend *extend) mc_plan = m->submits_3core; mc_count = m->n_submits_3core; } + /* FP16 transformer models (unified_act): single-core tasks only cover + * ~512 of 1024 spatial positions. The vendor runs all 3 NPU cores + * simultaneously (task_number=N*3) via kernel-level multi-core dispatch + * set up during rknn_init. openrknn's OWN mode doesn't do this kernel + * setup, so submitting with core_mask=0 still runs single-core. + * TODO #80: implement kernel-level multi-core dispatch for FP16. */ + int use_vendor_3core = 0; /* disabled: needs kernel integration */ if (mc_plan && mc_count > 0) { orknn_log(2, "run: multi-core submit (mask=0x%x, %u submits)",