Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 6 additions & 18 deletions openrknn/src/openrknn_drm.c
Original file line number Diff line number Diff line change
Expand Up @@ -288,7 +288,7 @@ int orknn_npu_submit(int fd, struct orknn_bo *task_bo,
popcount = 1;
}
if (effective == 0)
effective = 0x1; /* AUTO → core 0 */
effective = 0x1; /* AUTO → core 0 (kernel panics with 0 without vendor init) */

/* Dev: ORKNN_MAX_TASKS=N truncates every single-core submit to at
* most N tasks. Used for Phase-0 bisection of the ORKNN_ORACLE_PATCH
Expand Down Expand Up @@ -323,23 +323,11 @@ int orknn_npu_submit(int fd, struct orknn_bo *task_bo,
.core_mask = effective,
.fence_fd = -1,
.subcore_task = {
/* Single-core: kernel reads subcore_task[core_index] where
* core_index is the bit position of the active core. We
* populate all 3 single-core slots with the same range so
* mask=0x1, 0x2, 0x4 all work.
*
* Multi-core (experimental): slots [2..4] would need to
* carry per-core task regions from a multi-core task BO
* layout we don't yet parse. Filled with the same
* single-core range; kernel will either run all 3 cores
* on redundant work (wasteful + incorrect output) or
* error out. */
{ eff_task_start, eff_sc_count }, /* [0] mask=0x1 → core 0 */
{ eff_task_start, eff_sc_count }, /* [1] mask=0x2 → core 1 */
{ eff_task_start, eff_sc_count }, /* [2] mask=0x4 → core 2
* (and 3-core: core 0) */
{ eff_task_start, eff_sc_count }, /* [3] 3-core: core 1 */
{ eff_task_start, eff_sc_count }, /* [4] 3-core: core 2 */
{ eff_task_start, eff_sc_count },
{ eff_task_start, eff_sc_count },
{ eff_task_start, eff_sc_count },
{ eff_task_start, eff_sc_count },
{ eff_task_start, eff_sc_count },
},
};

Expand Down
18 changes: 10 additions & 8 deletions openrknn/src/openrknn_run.c
Original file line number Diff line number Diff line change
Expand Up @@ -2083,12 +2083,14 @@ int orknn_own_run(struct orknn_context *ctx, rknn_run_extend *extend)
mc_count = m->n_submits_3core;
}
/* FP16 transformer models (unified_act): single-core tasks only cover
* ~512 of 1024 spatial positions. The vendor runs all 3 NPU cores
* simultaneously (task_number=N*3) via kernel-level multi-core dispatch
* set up during rknn_init. openrknn's OWN mode doesn't do this kernel
* setup, so submitting with core_mask=0 still runs single-core.
* TODO #80: implement kernel-level multi-core dispatch for FP16. */
int use_vendor_3core = 0; /* disabled: needs kernel integration */
* ~512 of 1024 spatial positions. The vendor submits with task_number
* tripled (N*3) and core_mask from ctx, leaving subcore_task[] zeroed.
* The kernel auto-distributes tasks across all available cores. */
/* FP16 transformer models: single-core tasks cover ~512/1024 spatial
* positions. The vendor uses core_mask=0 (kernel auto-distributes
* 3x tasks across all 3 cores) with subcore_task zeroed. This requires
* kernel-level context setup that openrknn's OWN mode doesn't do.
* See #80 for the full multi-core dispatch investigation. */

if (mc_plan && mc_count > 0) {
orknn_log(2, "run: multi-core submit (mask=0x%x, %u submits)",
Expand Down Expand Up @@ -2146,8 +2148,8 @@ int orknn_own_run(struct orknn_context *ctx, rknn_run_extend *extend)
seg_copy.task_number -= trim;
}
}
int ret = orknn_npu_submit(ctx->npu_fd, &ctx->task_bo, &seg_copy,
ctx->core_mask);
int ret = orknn_npu_submit(ctx->npu_fd, &ctx->task_bo,
&seg_copy, ctx->core_mask);
if (ret) {
orknn_log(0, "run: segment %u submit failed", i);
return RKNN_ERR_FAIL;
Expand Down
Loading