* hex-mm: fix artificial limit in the solver that restricted number of act-prep threads
* hex-mm: fix warning
* hex-prof: do not apply --top to the timeline report
* hmx-mm: add suport for tiled act-processing to better distribute hvx work
* hex-l2: add tracing for l2flush events
* workqueue: redo the legacy workpool api to match hmx-queue and dma-queue
* hmx-mm: fix f32 activation buffer alignmnet for nhvx=5,6,7
* hex-work: minor cleanup for work-queue apis
* hex-work: further cleanup of the work-queue api
* hex-l2: optimize l2flushes at the opbatch level
* hex-work: remove unused mask
* hex-work: no need to drop hvx ctx in the work-queue
* hex-work: add explicit wakeup/suspend and make threads spin
* hex-bufs: mark any non-weight tensor as compute
* hex-dma: dma-queue support for alias queues and cached dma
* hex-l2: track tensor aliases and delay or skip flushes as much as possible
* hex-l2: simplify tensor alias handling
* hex-l2: handle overlapping views as a circular list of aliases
* hex-tens: add flags helper
* hex-l2: add helper for marking tensors clearn/dirty
* hex-l2: mark binary and rope outputs as l2-clean and keep the rest as is for now
* hex-l2: proper support for handling all tensor overlap scenarios
* hex-trace: instrument matmul init code and cleanup trace checks
* hex-thread: introduce dedicated main thread with explicit stack and priority
* hex-l2: track dirty state as bitmap and introduce threaded flush
* hex-trace: remove redundant checks for ctx != null
* hex-l2: allocate entire context as one buffer and l2fetch it after big flushes
* hex-l2: disable tensor clearing in binary and rope for now seems to cause issues with fusion
* hmx-mm: update act proc to use fastdivs and fix DMA overflow
* hmx-mm: make MUL_MAT_ID kernels robust to multi-chunk cases (start_row>0)
* hex-queue: remove obsolete queue interfaces and flush hmx-queue at the end of the op-batch
* hex-queue: dont use early wakeup for small op-batches
* hex-tensors: properly cap max_tensors in op-batches and dirty_map
* hex-l2: make sure threaded l2flush does proper rounding
* hex-l2: factor out htp_tensor_flush for reuse (if needed)
* hex-l2: optimize tensor flushes by coalescing flush-all
* hex-l2: optimize multi-threaded flush
* hex-drv: futureproof version checks
* hexagon: fix errors and warnings on windows
* hex-main: update main thread to only use dspqueue_read, dspqueue_peek is not available on some platforms
* hex-main: add fallback mode for dspqueue with callbacks
* hex-main: introduce fallback mode for using dspqueue callbacks for full op processing
* hex-main: remove early wakeup, not helping and seems to cause some errors with certain batch sizes
* hex-l2: make sure to use invalidate version of flushall
* hex-l2: dont try to trace early l2flush at the start of op-batch
* hex-main: remove offset_ctx that must be zero anyway
* hex-hmx: fix hmx_queue_depth to use idx_write - idx_read
* hex-hmx: use atomic_load for idx_read/write
* hex-main: add static assert to make sure n_threads are aligned
#include <algorithm>
#ifdef _WIN32
+# define WIN32_LEAN_AND_MEAN
+# ifndef NOMINMAX
+# define NOMINMAX
+# endif
+# include <windows.h>
# include <sal.h>
#else
# include <semaphore.h>
#endif
#pragma clang diagnostic ignored "-Wnested-anon-types"
+#pragma clang diagnostic ignored "-Wlanguage-extension-token"
#pragma clang diagnostic ignored "-Wgnu-anonymous-struct"
+#pragma clang diagnostic ignored "-Wmicrosoft-enum-value"
#include <AEEStdErr.h>
#include <dspqueue.h>
case HTP_TRACE_EVT_HVX_FA_K_PREP: return "HVX_K_PREP";
case HTP_TRACE_EVT_HVX_FA_V_PREP: return "HVX_V_PREP";
case HTP_TRACE_EVT_HMX_COMP: return "HMX_COMP";
+ case HTP_TRACE_EVT_L2FLUSH: return "L2FLUSH";
+ case HTP_TRACE_EVT_INIT: return "INIT";
default: return "UNKNOWN";
}
}
}
}
}
+
+ GGML_UNUSED(size);
}
// repack q4_0_tiled tensor into q4_0 data
}
}
}
+
+ GGML_UNUSED(size);
}
// repack q4_1 data into q4_1_tiled tensor
}
}
}
+
+ GGML_UNUSED(size);
}
// repack q4_1_tiled tensor into q4_1 data
}
}
}
+
+ GGML_UNUSED(size);
}
// repack q8_0 data into q8_0_tiled tensor
}
}
}
+
+ GGML_UNUSED(size);
}
// repack q8_0_tiled tensor into q8_0 data
}
}
}
+
+ GGML_UNUSED(size);
}
// repack mxfp4 data into mxfp4_tiled tensor
}
}
}
+
+ GGML_UNUSED(size);
}
// repack mxfp4_tiled tensor into mxfp4 data
}
}
}
+
+ GGML_UNUSED(size);
}
static void ggml_backend_hexagon_buffer_set_tensor(ggml_backend_buffer_t buffer,
static bool ggml_backend_hexagon_buffer_cpy_tensor(ggml_backend_buffer_t buffer,
const struct ggml_tensor * src,
struct ggml_tensor * dst) {
+ // we might optimize this later, for now take the slow path (ie get/set_tensor)
+ return false;
+
GGML_UNUSED(buffer);
GGML_UNUSED(src);
GGML_UNUSED(dst);
- // we might optimize this later, for now take the slow path (ie get/set_tensor)
- return false;
}
static void ggml_backend_hexagon_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) {
}
}
-static size_t ggml_backend_hexagon_buffer_type_get_alignment(ggml_backend_buffer_type_t buffer_type) {
+static size_t ggml_backend_hexagon_buffer_type_get_alignment(ggml_backend_buffer_type_t buft) {
return 128; // HVX alignment
- GGML_UNUSED(buffer_type);
+ GGML_UNUSED(buft);
}
static size_t ggml_backend_hexagon_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft, const struct ggml_tensor * t) {
return ggml_row_size(t->type, ne0) * ne1 * ne2 * ne3;
}
return ggml_nbytes(t);
+
+ GGML_UNUSED(buft);
}
-static size_t ggml_backend_hexagon_buffer_type_get_max_size(ggml_backend_buffer_type_t buffer_type) {
- auto * context = static_cast<ggml_backend_hexagon_buffer_type_context *>(buffer_type->context);
+static size_t ggml_backend_hexagon_buffer_type_get_max_size(ggml_backend_buffer_type_t buft) {
+ auto * context = static_cast<ggml_backend_hexagon_buffer_type_context *>(buft->context);
return context->sess->max_bufsize;
}
static bool ggml_backend_hexagon_buffer_type_is_host(ggml_backend_buffer_type_t buft) {
return opt_hostbuf;
+
GGML_UNUSED(buft);
}
static bool ggml_backend_hexagon_repack_buffer_type_is_host(ggml_backend_buffer_type_t buft) {
return false;
+
GGML_UNUSED(buft);
}
std::unordered_map<const ggml_tensor*, int> t_map; // tensor ptr to index
std::unordered_multimap<void*, int> d_map; // tensor data to index
+ struct tensor_range {
+ uint64_t start;
+ uint64_t end;
+ int bi;
+ std::vector<int> tensors;
+ };
+ std::vector<tensor_range> ranges;
+
unsigned int n_bufs; // num buffers in the batch
unsigned int n_tens; // num tensors ...
unsigned int n_ops; // num ops ...
b_map.clear();
t_map.clear();
d_map.clear();
+ ranges.clear();
}
ggml_hexagon_opbatch(ggml_hexagon_session *sess, size_t batch_size, size_t max_vmem) {
n_bufs_max = HTP_OP_MAX_BUFS;
n_ops_max = batch_size;
- n_tens_max = n_ops_max + n_ops_max * HTP_OP_MAX_INPUTS;
+ n_tens_max = std::min<size_t>(n_ops_max + n_ops_max * HTP_OP_MAX_INPUTS, HTP_OP_MAX_TENSORS);
b_vmem_max = max_vmem;
return bi;
}
+ void add_range(const htp_tensor * h, int ti) {
+ uint64_t t_start = h->data;
+ uint64_t t_end = t_start + h->size;
+ int bi = h->bi;
+
+ int first_match = -1;
+ int unused_idx = -1;
+ for (size_t i = 0; i < ranges.size(); i++) {
+ if (ranges[i].bi == -1) {
+ unused_idx = i;
+ continue;
+ }
+ if (ranges[i].bi != bi) {
+ continue;
+ }
+ if (ranges[i].start >= t_end || ranges[i].end <= t_start) {
+ continue;
+ }
+
+ if (first_match == -1) {
+ first_match = i;
+ HEX_VERBOSE("ggml-hex: %s range-grow #%d : bi %d [%p, %p) + #%d [%p, %p) -> [%p, %p)\n",
+ sess->c_name(), (int) i, ranges[i].bi,
+ (void *) (h_bufs[ranges[i].bi].base + ranges[i].start),
+ (void *) (h_bufs[ranges[i].bi].base + ranges[i].end),
+ ti,
+ (void *) (h_bufs[bi].base + t_start),
+ (void *) (h_bufs[bi].base + t_end),
+ (void *) (h_bufs[ranges[i].bi].base + std::min(ranges[i].start, t_start)),
+ (void *) (h_bufs[ranges[i].bi].base + std::max(ranges[i].end, t_end)));
+
+ ranges[i].start = std::min(ranges[i].start, t_start);
+ ranges[i].end = std::max(ranges[i].end, t_end);
+ ranges[i].tensors.push_back(ti);
+ } else {
+ HEX_VERBOSE("ggml-hex: %s range-merge #%d [%p, %p) + #%d [%p, %p) -> [%p, %p)\n",
+ sess->c_name(), first_match,
+ (void *) (h_bufs[bi].base + ranges[first_match].start),
+ (void *) (h_bufs[bi].base + ranges[first_match].end),
+ (int) i,
+ (void *) (h_bufs[bi].base + ranges[i].start),
+ (void *) (h_bufs[bi].base + ranges[i].end),
+ (void *) (h_bufs[bi].base + std::min(ranges[first_match].start, ranges[i].start)),
+ (void *) (h_bufs[bi].base + std::max(ranges[first_match].end, ranges[i].end)));
+
+ ranges[first_match].start = std::min(ranges[first_match].start, ranges[i].start);
+ ranges[first_match].end = std::max(ranges[first_match].end, ranges[i].end);
+ ranges[first_match].tensors.insert(
+ ranges[first_match].tensors.end(),
+ ranges[i].tensors.begin(),
+ ranges[i].tensors.end()
+ );
+ ranges[i].bi = -1;
+ }
+ }
+
+ if (first_match == -1) {
+ if (unused_idx != -1) {
+ ranges[unused_idx] = {t_start, t_end, bi, {ti}};
+ } else {
+ ranges.push_back({t_start, t_end, bi, {ti}});
+ }
+ }
+ }
+
bool same_shape(const htp_tensor * h, const ggml_tensor * t) const {
int64_t ne0 = t->ne[0];
int64_t ne1 = t->ne[1];
htp_tensor &h = h_tens[ti];
h.bi = add_buffer(sbuf);
+ h.ti = ti;
h.data = t_offset;
h.type = t->type;
h.nb[0] = t->nb[0]; h.nb[1] = t->nb[1]; h.nb[2] = t->nb[2]; h.nb[3] = t->nb[3];
}
+ h.alias = ti;
+ add_range(&h, ti);
+
h.flags = 0;
- if (ggml_backend_buffer_get_usage(t->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE) {
+ if (ggml_backend_buffer_get_usage(t->buffer) != GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {
h.flags |= HTP_TENSOR_COMPUTE;
}
o.dst[i] = (i < outputs.size() && outputs[i]) ? add_tensor(outputs[i]) : 0xffff;
}
}
+
+ void finalize_ranges() {
+ for (const auto & r : ranges) {
+ if (r.bi == -1) {
+ continue;
+ }
+ for (size_t i = 0; i < r.tensors.size(); i++) {
+ h_tens[r.tensors[i]].alias = r.tensors[(i + 1) % r.tensors.size()];
+ }
+ }
+ }
};
struct ggml_hexagon_opqueue {
void ggml_hexagon_session::flush_batch() {
if (op_batch->empty()) { return; }
+ op_batch->finalize_ranges();
+
htp_opbatch_req req {};
dspqueue_buffer dbuf{};
GGML_LOG_DEBUG("ggml-hex: %s allocating new session\n", this->name.c_str());
- domain * my_domain = get_domain(this->domain_id);
+ domain * my_domain = htpdrv_get_domain(this->domain_id);
if (my_domain == NULL) {
GGML_LOG_ERROR("ggml-hex: unable to get domain struct for CDSP\n");
throw std::runtime_error("ggml-hex: failed to get CDSP domain (see log for details)");
}
}
- if (opt_profile) {
- htp_iface_pmu_conf pmu_conf{};
- std::copy(opt_pmu_evt.begin(), opt_pmu_evt.end(), pmu_conf.events);
-
- err = htp_iface_profiler(this->handle, opt_profile, &pmu_conf);
- if (err != 0) {
- GGML_LOG_ERROR("ggml-hex: failed to enable profiling: 0x%08x\n", (unsigned) err);
- }
- }
-
// Allocate buffers and state for op batching
this->op_queue = new ggml_hexagon_opqueue(this, opt_opbatch, opt_opqueue);
throw std::runtime_error("ggml-hex: iface start failed (see log for details)");
}
this->valid_iface = true;
+
+ if (opt_profile) {
+ htp_iface_pmu_conf pmu_conf{};
+ std::copy(opt_pmu_evt.begin(), opt_pmu_evt.end(), pmu_conf.events);
+
+ err = htp_iface_profiler(this->handle, opt_profile, &pmu_conf);
+ if (err != 0) {
+ GGML_LOG_ERROR("ggml-hex: failed to enable profiling: 0x%08x\n", (unsigned) err);
+ }
+ }
}
void ggml_hexagon_session::release() noexcept(true) {
}
return true;
+
+ GGML_UNUSED(sinks);
}
static bool ggml_hexagon_precompute_flash_attn_params(
return false;
}
- GGML_UNUSED(sess);
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_matmul_is_hmx_eligible(
}
return true;
+
+ GGML_UNUSED(dst);
}
static bool ggml_hexagon_precompute_hmx_mm_params(
if (is_batched_val && wtype == GGML_TYPE_F16 && group_size > 1) {
// Try grouped path first
const bool use_dma_activation = (src1->nb[1]/sizeof(float) > (size_t)ne00_padded);
- size_t best_mblocks = SIZE_MAX;
- int best_act_threads = 0;
- size_t best_m_chunk = 0;
- size_t best_n_chunk = 0;
- size_t best_vtcm_size = 0;
-
- int act_threads = n_threads;
- while (act_threads >= 1) {
- const size_t f32_scratch_size = use_dma_activation ? hex_align_up(act_threads * HTP_MM_DMA_ACT_MULTIPLIER * ne00_padded * sizeof(float), HTP_MM_HMX_TILE_SIZE) : 0;
- size_t group_overhead = 256 + f32_scratch_size;
- size_t group_size_per_n, group_size_per_m, group_size_per_mn;
- htp_mm_hmx_get_batched_chunk_costs(ne00_padded, group_size, &group_size_per_n, &group_size_per_m, &group_size_per_mn);
-
- size_t m_chunk_candidate = 0;
- size_t n_chunk_candidate = 0;
- size_t vtcm_size_candidate = 0;
-
- if (htp_mm_hmx_compute_chunks(vtcm_budget, group_overhead, group_size_per_n, group_size_per_m, group_size_per_mn, hex_align_up(ne11, 32), ne01_padded,
- (size_t) ne01_padded * HTP_MM_HMX_COST_W_DEQUANT, (size_t) ne11 * HTP_MM_HMX_COST_A_CONVERT,
- &m_chunk_candidate, &n_chunk_candidate, &vtcm_size_candidate) == 0) {
- size_t exact_size = htp_mm_hmx_get_batched_vtcm_size(wtype, ne00_padded, m_chunk_candidate, n_chunk_candidate, group_size, use_dma_activation, pipeline, act_threads);
- if (exact_size <= vtcm_budget) {
- size_t mblocks = ((size_t) ne11 + m_chunk_candidate - 1) / m_chunk_candidate;
- if (mblocks < best_mblocks || (mblocks == best_mblocks && act_threads > best_act_threads)) {
- best_mblocks = mblocks;
- best_act_threads = act_threads;
- best_m_chunk = m_chunk_candidate;
- best_n_chunk = n_chunk_candidate;
- best_vtcm_size = exact_size;
- }
- }
- }
- if (act_threads == 1) {
- act_threads = 0;
- } else {
- act_threads /= 2;
- }
- }
-
- if (best_act_threads > 0) {
- m_chunk = best_m_chunk;
- n_chunk = best_n_chunk;
- vtcm_size = best_vtcm_size;
- act_threads_selected = best_act_threads;
+ if (htp_mm_hmx_solve_batched_params(wtype, ne00_padded, ne01_padded, ne11, group_size, use_dma_activation, n_threads, pipeline, vtcm_budget, &m_chunk, &n_chunk, &act_threads_selected, &vtcm_size)) {
use_grouped = true;
}
}
if (!use_grouped) {
// Fallback to simple 2D path (group_size = 1)
- size_t best_mblocks = SIZE_MAX;
- int best_act_threads = 0;
- size_t best_m_chunk = 0;
- size_t best_n_chunk = 0;
- size_t best_vtcm_size = 0;
-
- // For MUL_MAT_ID the kernel runs one 2D matmul per expert, with M equal to the number of rows routed to that expert.
- // A single expert can receive up to all routed rows (dst->ne[1]*dst->ne[2] = n_expert_used*n_tokens), so size the chunk
- // search for that upper bound rather than ne12 (token positions only).
- // We recompute m_chunk per expert against the actual count in the NPU kernel.
- const int m_id_rows = (int) ((size_t) dst->ne[1] * dst->ne[2]);
- const int m_for_chunks = is_matmul_id ? hex_align_up(m_id_rows, 32) : ne11_padded;
- const int m_for_cost = is_matmul_id ? m_id_rows : ne11;
-
- int act_threads = n_threads;
- while (act_threads >= 1) {
- const size_t act_f32_size = is_matmul_id ? 0 : hex_align_up(act_threads * HTP_MM_DMA_ACT_MULTIPLIER * ne00_padded * sizeof(float), HTP_MM_HMX_TILE_SIZE);
- size_t simple_2d_overhead = 256 + act_f32_size;
- size_t simple_2d_size_per_n, simple_2d_size_per_m, simple_2d_size_per_mn;
- htp_mm_hmx_get_2d_chunk_costs(wtype, ne00_padded, pipeline, aligned_tile_size, &simple_2d_size_per_n, &simple_2d_size_per_m, &simple_2d_size_per_mn);
-
- size_t m_chunk_candidate = 0;
- size_t n_chunk_candidate = 0;
- size_t vtcm_size_candidate = 0;
-
- if (htp_mm_hmx_compute_chunks(vtcm_budget, simple_2d_overhead, simple_2d_size_per_n, simple_2d_size_per_m, simple_2d_size_per_mn, m_for_chunks, ne01_padded,
- (size_t) ne01_padded * HTP_MM_HMX_COST_W_DEQUANT, (size_t) m_for_cost * HTP_MM_HMX_COST_A_CONVERT,
- &m_chunk_candidate, &n_chunk_candidate, &vtcm_size_candidate) == 0) {
- size_t exact_size = htp_mm_hmx_get_2d_vtcm_size(wtype, ne00_padded, m_chunk_candidate, n_chunk_candidate, pipeline, is_matmul_id ? 0 : act_threads, aligned_tile_size);
- if (exact_size <= vtcm_budget) {
- size_t mblocks = ((size_t) m_for_cost + m_chunk_candidate - 1) / m_chunk_candidate;
- if (mblocks < best_mblocks || (mblocks == best_mblocks && act_threads > best_act_threads)) {
- best_mblocks = mblocks;
- best_act_threads = act_threads;
- best_m_chunk = m_chunk_candidate;
- best_n_chunk = n_chunk_candidate;
- best_vtcm_size = exact_size;
- }
- }
- }
- if (act_threads == 1) {
- act_threads = 0;
- } else {
- act_threads /= 2;
- }
- }
-
- if (best_act_threads > 0) {
- m_chunk = best_m_chunk;
- n_chunk = best_n_chunk;
- vtcm_size = best_vtcm_size;
- act_threads_selected = best_act_threads;
- } else {
+ const int m_id_rows = (int) ((size_t) dst->ne[1] * dst->ne[2]);
+ if (!htp_mm_hmx_solve_2d_params(wtype, ne00_padded, m_id_rows, ne01_padded, ne11_padded, ne11, n_threads, pipeline, is_matmul_id, aligned_tile_size, vtcm_budget, &m_chunk, &n_chunk, &act_threads_selected, &vtcm_size)) {
return false;
}
}
kparams->src1_row_size = (wtype == GGML_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10);
kparams->vtcm_size = vtcm_size;
kparams->vtcm_src0_size = 0;
+ kparams->div_n_act_threads = init_fastdiv_values(act_threads_selected);
+ kparams->div_ne00_padded = init_fastdiv_values(ne00_padded);
kparams->vtcm_src1_size = 0;
kparams->vtcm_dst_size = 0;
kparams->kernel_type = HTP_MM_KERNEL_HMX_2D;
}
return true;
+
+ GGML_UNUSED(src0);
}
static void ggml_hexagon_precompute_hvx_mm_params(
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_add_id(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_unary(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_sum_rows(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
}
return true;
+
+ GGML_UNUSED(sess);
}
-static bool ggml_hexagon_supported_activations(const struct ggml_hexagon_session * sess,
- const struct ggml_tensor * op) {
+static bool ggml_hexagon_supported_activations(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
const struct ggml_tensor * src0 = op->src[0];
const struct ggml_tensor * src1 = op->src[1];
const struct ggml_tensor * dst = op;
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_softmax(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_set_rows(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_get_rows(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_argsort(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_rope(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
return false;
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_ssm_conv(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
}
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_pad(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
return false;
}
- GGML_UNUSED(sess);
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_cumsum(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
return false;
}
- GGML_UNUSED(sess);
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_diag(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
return false;
}
- GGML_UNUSED(sess);
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_solve_tri(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
return false;
}
- GGML_UNUSED(sess);
return true;
+
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_tri(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
}
}
}
+
+ GGML_UNUSED(backend);
}
static struct ggml_backend_i hexagon_backend_i = {
}
static bool ggml_hexagon_supported_cpy(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
+ GGML_UNUSED(sess);
+
const struct ggml_tensor * src0 = op->src[0];
const struct ggml_tensor * dst = op;
}
return true;
+ GGML_UNUSED(sess);
}
static bool ggml_hexagon_supported_fill(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) {
return false;
}
- GGML_UNUSED(sess);
return true;
+ GGML_UNUSED(sess);
}
static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, const struct ggml_tensor * op) {
}
return NULL;
+ GGML_UNUSED(reg);
}
template<typename T> std::vector<T> str_to_vec(const char* str) {
// Init Arch first since it affects other defaults
if (!str_arch) {
- int err = get_hex_arch_ver(CDSP_DOMAIN_ID, &opt_arch);
+ int err = htpdrv_get_arch(CDSP_DOMAIN_ID, &opt_arch);
if (err != 0) {
GGML_LOG_ERROR("ggml-hex: failed to query HTP version (err %d) defaulting to v73\n", err);
opt_arch = 73;
+ } else {
+ if (opt_arch < 73) {
+ GGML_LOG_WARN("ggml-hex: Hexagon arch v%d is under supported range, capping at v73\n", opt_arch);
+ opt_arch = 73;
+ } else if (opt_arch > 81) {
+ GGML_LOG_WARN("ggml-hex: Hexagon arch v%d is over supported range, capping at v81\n", opt_arch);
+ opt_arch = 81;
+ }
}
} else {
if (str_arch[0] == 'v' || str_arch[0] == 'V') {
-// sample drv interface
-
-#pragma clang diagnostic ignored "-Wgnu-anonymous-struct"
-#pragma clang diagnostic ignored "-Wmissing-prototypes"
-#pragma clang diagnostic ignored "-Wsign-compare"
-
#include <filesystem>
#include <set>
#include <sstream>
#include <string>
+
#ifdef _WIN32
# define WIN32_LEAN_AND_MEAN
# ifndef NOMINMAX
# include <windows.h>
# include <winevt.h>
#else
-# include <dlfcn.h>
-# include <unistd.h>
+# include <dlfcn.h>
+# include <unistd.h>
#endif
+
+#pragma clang diagnostic ignored "-Wgnu-anonymous-struct"
+#pragma clang diagnostic ignored "-Wmissing-prototypes"
+#pragma clang diagnostic ignored "-Wsign-compare"
+#pragma clang diagnostic ignored "-Wlanguage-extension-token"
+#pragma clang diagnostic ignored "-Wmicrosoft-enum-value"
+#pragma clang diagnostic ignored "-Wnested-anon-types"
+
#include "ggml-impl.h"
#include "htp-drv.h"
#include "libdl.h"
return AEE_SUCCESS;
}
-domain * get_domain(int domain_id) {
+domain * htpdrv_get_domain(int domain_id) {
int i = 0;
int size = sizeof(supported_domains) / sizeof(domain);
return NULL;
}
-int get_hex_arch_ver(int domain, int * arch) {
+int htpdrv_get_arch(int domain, int * arch) {
if (!remote_handle_control_pfn) {
GGML_LOG_ERROR("ggml-hex: remote_handle_control is not supported on this device\n");
return AEE_EUNSUPPORTEDAPI;
return err;
}
- switch (arch_ver.capability & 0xff) {
- case 0x68:
- *arch = 68;
- return 0;
- case 0x69:
- *arch = 69;
- return 0;
- case 0x73:
- *arch = 73;
- return 0;
- case 0x75:
- *arch = 75;
- return 0;
- case 0x79:
- *arch = 79;
- return 0;
- case 0x81:
- *arch = 81;
- return 0;
- }
- return -1;
+ uint32_t val = arch_ver.capability & 0xff;
+ *arch = (int) ((val >> 4) * 10 + (val & 0x0f));
+ return 0;
}
HTPDRV_API int htpdrv_init(void);
/**
- * get_domain API: get domain struct from domain value.
+ * htpdrv_get_domain API: get domain struct from domain value.
*
* @param[in] domain value of a domain
* @return Returns domain struct of the domain if it is supported or else
* returns NULL.
*
*/
-HTPDRV_API domain * get_domain(int domain_id);
+HTPDRV_API domain * htpdrv_get_domain(int domain_id);
/**
- * get_hex_arch_ver API: query the Hexagon processor architecture version information
+ * htpdrv_get_arch API: query the Hexagon processor architecture version information
*
* @param[in] domain_id value of a domain
* @param[out] Arch version (73, 75, ...)
* non-zero if error, return value points to the error.
*
*/
-HTPDRV_API int get_hex_arch_ver(int domain, int * arch);
+HTPDRV_API int htpdrv_get_arch(int domain, int * arch);
#ifdef __cplusplus
}
add_library(${HTP_LIB} SHARED
main.c
htp_iface_skel.c
- worker-pool.c
- hex-dma.c
+ work-queue.c
+ dma-queue.c
hmx-queue.c
+ htp-tensor.c
+ matmul-ops.c
+ flash-attn-ops.c
gated-delta-net-ops.c
binary-ops.c
unary-ops.c
diag-ops.c
solve-tri-ops.c
pad-ops.c
- flash-attn-ops.c
- matmul-ops.c
argsort-ops.c
)
#include "htp-ctx.h"
#include "htp-ops.h"
#include "htp-ops.h"
+#include "htp-tensor.h"
#define htp_act_preamble \
const struct htp_tensor * src0 = actx->octx->src[0]; \
}
int op_activations(struct htp_ops_context * octx) {
- int err = HTP_STATUS_OK;
-
switch (octx->src[0]->type) {
case HTP_TYPE_F32:
- err = execute_op_activations_f32(octx);
- break;
+ return execute_op_activations_f32(octx);
default:
- err = HTP_STATUS_NO_SUPPORT;
- break;
+ return HTP_STATUS_NO_SUPPORT;
}
-
- return err;
}
int32_t * indices_buf = (int32_t *) (spad + values_size); \
uint32_t nb01 = src0->nb[1]; \
uint32_t nb1 = dst->nb[1]; \
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[i] : NULL; \
+ struct htp_thread_trace * tr = &octx->ctx->trace[i]; \
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, start_row); \
for (uint32_t r = start_row; r < end_row; r++) { \
uint32_t src_offset = r * nb01; \
const HVX_Vector ind_init_vec = *(HVX_Vector *)argosrt_ramp_lut;
const HVX_Vector ind_diff_vec = Q6_V_vsplat_R(32);
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[i] : NULL;
+ struct htp_thread_trace * tr = &octx->ctx->trace[i];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, start_row);
for (uint32_t r = start_row; r < end_row; r++) {
#include "htp-ctx.h"
#include "htp-ops.h"
#include "htp-ops.h"
+#include "htp-tensor.h"
#ifndef MIN
#define MIN(a, b) ((a) < (b) ? (a) : (b))
#include "ggml-common.h"
#include "htp-ctx.h"
#include "htp-ops.h"
+#include "htp-tensor.h"
#include "hvx-types.h"
#include "hvx-utils.h"
#include "hex-dma.h"
int op_cumsum(struct htp_ops_context * octx) {
const struct htp_tensor * dst = octx->dst;
- int err = HTP_STATUS_OK;
-
switch (dst->type) {
case HTP_TYPE_F32:
- err = op_cumsum_f32(octx);
- break;
+ return op_cumsum_f32(octx);
default:
- err = HTP_STATUS_NO_SUPPORT;
- break;
+ return HTP_STATUS_NO_SUPPORT;
}
-
- return err;
}
--- /dev/null
+#include "dma-queue.h"
+
+#include <stdbool.h>
+#include <stdlib.h>
+#include <string.h>
+
+#pragma clang diagnostic ignored "-Wunused-function"
+
+static inline uint32_t pow2_ceil(uint32_t x) {
+ if (x <= 1) {
+ return 1;
+ }
+ int p = 2;
+ x--;
+ while (x >>= 1) {
+ p <<= 1;
+ }
+ return p;
+}
+
+static inline uintptr_t align_up(uintptr_t addr, size_t align) {
+ return (addr + align - 1) & ~(align - 1);
+}
+
+size_t dma_queue_sizeof(size_t capacity) {
+ capacity = pow2_ceil(capacity);
+
+ size_t size_q = sizeof(dma_queue);
+ size_t offset_r = align_up(size_q, HEX_L2_LINE_SIZE);
+ size_t size_r = sizeof(dma_ring);
+ size_t offset_desc = align_up(offset_r + size_r, HEX_L2_LINE_SIZE);
+ size_t size_desc = capacity * sizeof(dma_descriptor_2d);
+ size_t offset_dptr = align_up(offset_desc + size_desc, HEX_L2_LINE_SIZE);
+ size_t size_dptr = capacity * sizeof(dma_ptr);
+
+ return offset_dptr + size_dptr;
+}
+
+size_t dma_queue_alignof(void) {
+ return HEX_L2_LINE_SIZE;
+}
+
+dma_queue_t dma_queue_init(void * ptr, size_t capacity, uintptr_t vtcm_base, size_t vtcm_size, struct htp_thread_trace * trace) {
+ capacity = pow2_ceil(capacity);
+
+ size_t size_q = sizeof(dma_queue);
+ size_t offset_r = align_up(size_q, HEX_L2_LINE_SIZE);
+ size_t size_r = sizeof(dma_ring);
+ size_t offset_desc = align_up(offset_r + size_r, HEX_L2_LINE_SIZE);
+ size_t size_desc = capacity * sizeof(dma_descriptor_2d);
+ size_t offset_dptr = align_up(offset_desc + size_desc, HEX_L2_LINE_SIZE);
+ size_t size_dptr = capacity * sizeof(dma_ptr);
+
+ size_t total_size = offset_dptr + size_dptr;
+ memset(ptr, 0, total_size);
+
+ dma_queue * q = (dma_queue *) ptr;
+ dma_ring * r = (dma_ring *) ((uintptr_t) ptr + offset_r);
+
+ q->ring = r;
+ q->nocache = 0;
+ q->alias = false;
+
+ r->trace = trace;
+ r->vtcm_base = vtcm_base;
+ r->vtcm_end = vtcm_base + vtcm_size;
+ r->capacity = capacity;
+ r->idx_mask = capacity - 1;
+ r->push_idx = 0;
+ r->pop_idx = 0;
+
+ r->desc = (dma_descriptor_2d *) ((uintptr_t) ptr + offset_desc);
+ r->dptr = (dma_ptr *) ((uintptr_t) ptr + offset_dptr);
+ r->tail = &r->desc[capacity - 1];
+
+ FARF(HIGH, "dma-queue: capacity %u, unified memory size %zu\n", capacity, total_size);
+
+ return q;
+}
+
+void dma_queue_free(dma_queue_t q) {
+ (void) q;
+}
+
+size_t dma_queue_alias_sizeof(void) {
+ return sizeof(dma_queue);
+}
+
+dma_queue_t dma_queue_alias_init(void * ptr, dma_queue_t main_q, uint8_t nocache) {
+ dma_queue * q = (dma_queue *) ptr;
+ memset(q, 0, sizeof(dma_queue));
+
+ q->ring = main_q->ring;
+ q->nocache = nocache;
+ q->alias = true;
+
+ return q;
+}
+
+void dma_queue_alias_free(dma_queue_t q) {
+ (void) q;
+}
+
+void dma_queue_flush(dma_queue_t q) {
+ while (dma_queue_pop(q).dst != NULL) ;
+}
--- /dev/null
+#ifndef HTP_DMA_H
+#define HTP_DMA_H
+
+#include <HAP_farf.h>
+#include <hexagon_types.h>
+#include <stdbool.h>
+#include <stdint.h>
+#include "hex-utils.h"
+
+#include "hex-profile.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+// Define the HW descriptor structs here since the ones in HexSDK are a bit out of date
+typedef struct dma_descriptor_1d_s {
+ void * next;
+ uint32_t size:24;
+ uint32_t desc_size:2;
+ uint32_t dst_comp:1;
+ uint32_t src_comp:1;
+ uint32_t dst_bypass:1;
+ uint32_t src_bypass:1;
+ uint32_t order:1;
+ uint32_t done:1;
+ void * src;
+ void * dst;
+} dma_descriptor_1d;
+
+#if __HVX_ARCH__ < 75
+
+typedef struct dma_descriptor_2d_s {
+ void * next;
+ uint32_t reserved0:24;
+ uint32_t desc_size:2;
+ uint32_t dst_comp:1;
+ uint32_t src_comp:1;
+ uint32_t dst_bypass:1;
+ uint32_t src_bypass:1;
+ uint32_t order:1;
+ uint32_t done:1;
+ void * src;
+ void * dst;
+ uint32_t desc_type:8;
+ uint32_t reserved1:24;
+ uint32_t row_size:16;
+ uint32_t nrows:16;
+ uint32_t src_stride:16;
+ uint32_t dst_stride:16;
+ uint32_t src_offset:16;
+ uint32_t dst_offset:16;
+} dma_descriptor_2d;
+
+#else
+
+typedef struct dma_descriptor_2d_s {
+ void * next;
+ uint32_t dst_stride:24;
+ uint32_t desc_size:2;
+ uint32_t dst_comp:1;
+ uint32_t src_comp:1;
+ uint32_t dst_bypass:1;
+ uint32_t src_bypass:1;
+ uint32_t order:1;
+ uint32_t done:1;
+ void * src;
+ void * dst;
+ uint32_t desc_type:8;
+ uint32_t reserved0:24;
+ uint32_t row_size:24;
+ uint32_t nrows_lo:8;
+ uint32_t nrows_hi:8;
+ uint32_t src_stride:24;
+ uint32_t offset:24;
+ uint32_t reserved1:8;
+} dma_descriptor_2d;
+
+#endif
+
+typedef struct {
+ void *dst;
+ const void *src;
+} dma_ptr;
+
+typedef struct dma_ring_s dma_ring;
+struct dma_ring_s {
+ dma_descriptor_2d * desc; // descriptor pointers
+ dma_descriptor_2d * tail; // tail pointer
+ dma_ptr * dptr; // dst/src pointers
+ uint32_t push_idx;
+ uint32_t pop_idx;
+ uint32_t capacity;
+ uint32_t idx_mask;
+ struct htp_thread_trace * trace;
+ uintptr_t vtcm_base;
+ uintptr_t vtcm_end;
+};
+
+typedef struct dma_queue_s dma_queue;
+typedef dma_queue * dma_queue_t;
+
+struct dma_queue_s {
+ dma_ring * ring; // Points to the descriptor ring state
+ uint8_t nocache; // Queue-specific bypass flag
+ bool alias; // When set, dma_queue_delete will not free the ring
+};
+
+void dma_queue_flush(dma_queue_t q);
+
+size_t dma_queue_sizeof(size_t capacity);
+size_t dma_queue_alignof(void);
+dma_queue_t dma_queue_init(void * ptr, size_t capacity, uintptr_t vtcm_base, size_t vtcm_size, struct htp_thread_trace * trace);
+void dma_queue_free(dma_queue_t q);
+
+size_t dma_queue_alias_sizeof(void);
+dma_queue_t dma_queue_alias_init(void * ptr, dma_queue_t main_q, uint8_t nocache);
+void dma_queue_alias_free(dma_queue_t q);
+
+// TODO: technically we don't need these and could use Q6_dmstart/wait/etc instead
+// but those do not seem to always compiler properly.
+static inline void dmstart(void * next) {
+ asm volatile(" release(%0):at" : : "r"(next));
+ asm volatile(" dmstart(%0)" : : "r"(next));
+}
+
+static inline void dmlink(void * cur, void * next) {
+ asm volatile(" release(%0):at" : : "r"(next));
+ asm volatile(" dmlink(%0, %1)" : : "r"(cur), "r"(next));
+}
+
+static inline unsigned int dmpoll(void) {
+ unsigned int ret = 0;
+ asm volatile(" %0 = dmpoll" : "=r"(ret) : : "memory");
+ return ret;
+}
+
+static inline unsigned int dmwait(void) {
+ unsigned int ret = 0;
+ asm volatile(" %0 = dmwait" : "=r"(ret) : : "memory");
+ return ret;
+}
+
+static inline dma_ptr dma_make_ptr(void *dst, const void *src)
+{
+ dma_ptr p = { dst, src };
+ return p;
+}
+
+static inline bool dma_is_vtcm(const dma_queue * q, const void * ptr) {
+ return (uintptr_t) ptr >= q->ring->vtcm_base && (uintptr_t) ptr < q->ring->vtcm_end;
+}
+
+static inline bool dma_queue_push_single_1d(dma_queue * q, dma_ptr dptr, size_t size) {
+ dma_ring * r = q->ring;
+ if (((r->push_idx + 1) & r->idx_mask) == r->pop_idx) {
+ FARF(HIGH, "dma-push: queue full\n");
+ return false;
+ }
+
+ dma_descriptor_1d * desc = (dma_descriptor_1d *) &r->desc[r->push_idx];
+ desc->src = (void *) dptr.src;
+ desc->dst = (void *) dptr.dst;
+ desc->size = size;
+
+ r->dptr[r->push_idx] = dptr;
+
+ if (size) {
+ desc->next = NULL;
+ desc->desc_size = 0; // 1D mode
+ desc->src_bypass = dma_is_vtcm(q, dptr.src) ? 1 : q->nocache;
+ desc->dst_bypass = dma_is_vtcm(q, dptr.dst) ? 1 : q->nocache;
+ desc->order = 0;
+ desc->done = 0;
+
+ htp_trace_event_start(r->trace, HTP_TRACE_EVT_DMA, r->push_idx);
+ dmlink(r->tail, desc);
+ r->tail = (dma_descriptor_2d *) desc;
+ } else {
+ desc->desc_size = 0;
+ desc->done = 1;
+ }
+
+ r->push_idx = (r->push_idx + 1) & r->idx_mask;
+ return true;
+}
+
+static inline bool dma_queue_push_single_2d(dma_queue * q, dma_ptr dptr, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) {
+ dma_ring * r = q->ring;
+ if (((r->push_idx + 1) & r->idx_mask) == r->pop_idx) {
+ FARF(HIGH, "dma-push: queue full\n");
+ return false;
+ }
+
+ dma_descriptor_2d * desc = &r->desc[r->push_idx];
+
+ desc->next = NULL;
+ desc->reserved0 = 0;
+ desc->reserved1 = 0;
+ desc->desc_size = 1; // 2d mode
+ desc->src_bypass = dma_is_vtcm(q, dptr.src) ? 1 : q->nocache;
+ desc->dst_bypass = dma_is_vtcm(q, dptr.dst) ? 1 : q->nocache;
+ desc->src_comp = 0;
+ desc->dst_comp = 0;
+ desc->order = 0;
+ desc->done = 0;
+ desc->src_stride = src_stride;
+ desc->dst_stride = dst_stride;
+ desc->src = (void *) dptr.src;
+ desc->dst = (void *) dptr.dst;
+ desc->row_size = row_size;
+
+#if __HVX_ARCH__ < 75
+ desc->desc_type = 0; // 2d (16-bit) mode
+ desc->nrows = nrows;
+ desc->src_offset = 0;
+ desc->dst_offset = 0;
+#else
+ desc->desc_type = 9; // 2d (24-bit) mode
+ desc->nrows_lo = (nrows & 0xff);
+ desc->nrows_hi = (nrows >> 8);
+ desc->offset = 0;
+#endif
+
+ r->dptr[r->push_idx] = dptr;
+
+ if (nrows) {
+ htp_trace_event_start(r->trace, HTP_TRACE_EVT_DMA, r->push_idx);
+ dmlink(r->tail, desc);
+ r->tail = desc;
+ } else {
+ desc->done = 1;
+ }
+
+ r->push_idx = (r->push_idx + 1) & r->idx_mask;
+ return true;
+}
+
+static inline dma_ptr dma_queue_pop(dma_queue * q) {
+ dma_ring * r = q->ring;
+ dma_ptr dptr = { NULL };
+
+ if (r->push_idx == r->pop_idx) {
+ return dptr;
+ }
+
+ dma_descriptor_2d * desc = &r->desc[r->pop_idx];
+
+ // Wait for desc to complete
+ if (!desc->done) {
+ while (!desc->done) {
+ dmpoll();
+ }
+ }
+ htp_trace_event_stop(r->trace, HTP_TRACE_EVT_DMA, r->pop_idx);
+
+ dptr = r->dptr[r->pop_idx];
+
+ r->pop_idx = (r->pop_idx + 1) & r->idx_mask;
+ return dptr;
+}
+
+static inline dma_ptr dma_queue_pop_nowait(dma_queue * q) {
+ dma_ring * r = q->ring;
+ dma_ptr dptr = { NULL };
+
+ if (r->push_idx == r->pop_idx) {
+ return dptr;
+ }
+
+ dptr = r->dptr[r->pop_idx];
+
+ r->pop_idx = (r->pop_idx + 1) & r->idx_mask;
+ return dptr;
+}
+
+static inline bool dma_queue_empty(dma_queue * q) {
+ return q->ring->push_idx == q->ring->pop_idx;
+}
+
+static inline uint32_t dma_queue_depth(dma_queue * q) {
+ return (q->ring->push_idx - q->ring->pop_idx) & q->ring->idx_mask;
+}
+
+static inline uint32_t dma_queue_capacity(dma_queue * q) {
+ return q->ring->capacity;
+}
+
+#if __HVX_ARCH__ < 75
+
+// Overflow-safe DMA push: all 2d descriptor fields (row_size, nrows, src_stride, dst_stride) are 16-bit, max 65535.
+// This version transparently handles values that exceed the 16-bit limit and submits chained DMA transtions.
+
+#define DMA_MAX_FIELD_VAL 65535u
+
+static inline bool dma_queue_push(dma_queue *q, dma_ptr dptr, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) {
+ // Fast path: everything fits in 16 bits
+ if (nrows == 0 || __builtin_expect(
+ row_size <= DMA_MAX_FIELD_VAL &&
+ nrows <= DMA_MAX_FIELD_VAL &&
+ src_stride <= DMA_MAX_FIELD_VAL &&
+ dst_stride <= DMA_MAX_FIELD_VAL, 1)) {
+ return dma_queue_push_single_2d(q, dptr, dst_stride, src_stride, row_size, nrows);
+ }
+
+ // Contiguous block
+ // Use 1d DMA mode which supports sizes up to 24-bits (16MB)
+ if (nrows == 1 || (row_size == src_stride && row_size == dst_stride)) {
+ size_t total = row_size * nrows;
+ return dma_queue_push_single_1d(q, dptr, total);
+ }
+
+ // Stride overflow - fall back to row-by-row.
+ {
+ const uint8_t *src = (const uint8_t *) dptr.src;
+ uint8_t *dst = (uint8_t *) dptr.dst;
+ for (size_t r = 0; r < nrows; ++r) {
+ dma_ptr p = dma_make_ptr(dst + r * dst_stride, src + r * src_stride);
+ if (!dma_queue_push_single_1d(q, p, row_size))
+ return false;
+ if (r + 1 < nrows)
+ dma_queue_pop(q);
+ }
+ return true;
+ }
+}
+
+#else // HVX_ARCH >= 75
+
+static inline bool dma_queue_push(dma_queue *q, dma_ptr dptr, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) {
+ // On v75 and up we always use 2d 24-bit mode
+ return dma_queue_push_single_2d(q, dptr, dst_stride, src_stride, row_size, nrows);
+}
+
+#endif
+
+static inline bool dma_queue_push_ddr_to_vtcm(dma_queue * q, dma_ptr dptr, size_t dst_row_size, size_t src_row_size, size_t nrows) {
+ return dma_queue_push(q, dptr, dst_row_size, src_row_size, src_row_size, nrows);
+}
+
+static inline bool dma_queue_push_vtcm_to_ddr(dma_queue * q, dma_ptr dptr, size_t dst_row_size, size_t src_row_size, size_t nrows) {
+ return dma_queue_push(q, dptr, dst_row_size, src_row_size, dst_row_size, nrows);
+}
+
+#define DMA_CACHE_MAX_SIZE 256U
+
+typedef struct {
+ uint8_t *base;
+ uint32_t line_size;
+ uint32_t capacity;
+ uint32_t src[DMA_CACHE_MAX_SIZE];
+ uint16_t age[DMA_CACHE_MAX_SIZE];
+} dma_cache;
+
+static inline void dma_cache_init(dma_cache *c, uint8_t *base, uint32_t line_size, uint32_t capacity)
+{
+ c->capacity = (capacity > DMA_CACHE_MAX_SIZE) ? DMA_CACHE_MAX_SIZE : capacity;
+ c->base = base;
+ c->line_size = line_size;
+
+ for (unsigned i=0; i < c->capacity; i++) {
+ c->src[i] = 0;
+ c->age[i] = 0;
+ }
+}
+
+static inline bool dma_cache_push(dma_queue *q, dma_cache *c, const uint8_t * src, uint32_t dst_stride, uint32_t src_stride, uint32_t row_size, uint32_t nrows)
+{
+ uint32_t o_idx = 0;
+ uint16_t o_age = 0;
+ uint8_t * dst = 0;
+
+ for (unsigned i=0; i < c->capacity; i++) {
+ if (c->src[i] == (uint32_t) src) {
+ c->age[i] = 0;
+ dst = c->base + (i * c->line_size); nrows = 0; // dummy dma
+ } else {
+ c->age[i]++;
+ if (c->age[i] > o_age) { o_age = c->age[i]; o_idx = i; }
+ }
+ }
+ if (!dst) {
+ c->age[o_idx] = 0;
+ c->src[o_idx] = (uint32_t) src;
+ dst = c->base + o_idx * c->line_size; // normal nrows dma
+ return dma_queue_push(q, dma_make_ptr(dst, src), dst_stride, src_stride, row_size, nrows);
+ }
+
+ return dma_queue_push_single_1d(q, dma_make_ptr(dst, src), 0);
+}
+
+#ifdef __cplusplus
+} // extern "C"
+#endif
+
+#endif /* HTP_DMA_H */
#include "hvx-reduce.h"
#include "hvx-flash-attn.h"
#include "htp-vtcm.h"
-#include "worker-pool.h"
+#include "work-queue.h"
#define GGML_COMMON_DECL_C
#include "ggml-common.h"
if (ir0 >= ir1) return;
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL;
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith];
dma_queue * dma = octx->ctx->dma[ith];
return;
}
- struct htp_thread_trace * tr = factx->octx->ctx ? &factx->octx->ctx->trace[i] : NULL;
+ struct htp_thread_trace * tr = &factx->octx->ctx->trace[i];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_K_PREP, (uint16_t) (args->kv_start + start));
hmx_interleave_rows_to_tiles(factx->vtcm_k_tiles, (const __fp16 *) args->curr_k, total_rows, factx->DK,
args->src_stride, start, end);
}
static void fa_phase_k_interleave(struct hmx_fa_context * factx, uint32_t kv_rows, size_t src_stride, void * curr_k, uint32_t kv_start) {
- worker_pool_context_t wp = factx->octx->ctx->worker_pool;
+ work_queue_t wp = factx->octx->ctx->work_queue;
uint32_t n = 1;
if (factx->n_threads > 1 && kv_rows >= factx->n_threads * 2) {
n = factx->n_threads;
uint32_t rows_per_t = hex_align_up(hmx_ceil_div(kv_rows, n), 2);
fa_k_int_args_t args = { factx, kv_rows, src_stride, curr_k, kv_start, rows_per_t };
if (n > 1) {
- worker_pool_run_func(wp, fa_k_interleave_thread, &args, n);
+ work_queue_run(wp, fa_k_interleave_thread, &args, n);
} else {
fa_k_interleave_thread(1, 0, &args);
}
__fp16 * v_tiles_dst = (__fp16 *) args->v_tiles_dst;
- struct htp_thread_trace * tr = factx->octx->ctx ? &factx->octx->ctx->trace[i] : NULL;
+ struct htp_thread_trace * tr = &factx->octx->ctx->trace[i];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_V_PREP, (uint16_t) (args->kv_start + start));
hmx_interleave_cols_to_tiles(v_tiles_dst, (const __fp16 *) args->v_src, total_rows, factx->DV,
args->src_stride, (uint32_t) args->n_col_tiles, start, end);
void * v_tiles_dst,
size_t n_col_tiles,
uint32_t kv_start) {
- worker_pool_context_t wp = factx->octx->ctx->worker_pool;
+ work_queue_t wp = factx->octx->ctx->work_queue;
uint32_t n = 1;
if (factx->n_threads > 1 && kv_rows >= factx->n_threads * 2) {
n = factx->n_threads;
uint32_t rows_per_t = hex_align_up(hmx_ceil_div(kv_rows, n), 2);
fa_v_int_args_t args = { factx, kv_rows, src_stride, v_src, v_tiles_dst, n_col_tiles, kv_start, rows_per_t };
if (n > 1) {
- worker_pool_run_func(wp, fa_v_interleave_thread, &args, n);
+ work_queue_run(wp, fa_v_interleave_thread, &args, n);
} else {
fa_v_interleave_thread(1, 0, &args);
}
const size_t start = (size_t) i * rows_per_t;
const size_t end = hex_smin(start + rows_per_t, factx->g_br);
- struct htp_thread_trace * tr = factx->octx->ctx ? &factx->octx->ctx->trace[i] : NULL;
+ struct htp_thread_trace * tr = &factx->octx->ctx->trace[i];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_Q_PREP, (uint16_t) (args->q_start * G + start));
// Parallel initialization of per-block state
uint32_t kv_head,
uint32_t ib3,
size_t n_rows_g) {
- worker_pool_context_t wp = factx->octx->ctx->worker_pool;
+ work_queue_t wp = factx->octx->ctx->work_queue;
uint32_t n = 1;
if (factx->n_threads > 1 && n_rows_g >= (size_t) (factx->n_threads * 2)) {
n = factx->n_threads;
args.q_transposed = q->nb[1] < q->nb[2];
atomic_init(&args.barrier, n);
if (n > 1) {
- worker_pool_run_func(wp, fa_q_load_thread, &args, n);
+ work_queue_run(wp, fa_q_load_thread, &args, n);
} else {
fa_q_load_thread(1, 0, &args);
}
return;
}
- struct htp_thread_trace * tr = factx->octx->ctx ? &factx->octx->ctx->trace[i] : NULL;
+ struct htp_thread_trace * tr = &factx->octx->ctx->trace[i];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_O_PROC, (uint16_t) (args->q_start * G + start));
const struct htp_tensor * dst = args->dst;
return;
}
- struct htp_thread_trace * tr = factx->octx->ctx ? &factx->octx->ctx->trace[i] : NULL;
+ struct htp_thread_trace * tr = &factx->octx->ctx->trace[i];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_O_PROC, (uint16_t) (args->q_start * G + start));
const struct htp_tensor * dst = args->dst;
uint32_t kv_head,
uint32_t ib3,
size_t n_rows_g) {
- worker_pool_context_t wp = factx->octx->ctx->worker_pool;
+ work_queue_t wp = factx->octx->ctx->work_queue;
uint32_t n = 1;
if (factx->n_threads > 1 && n_rows_g >= (size_t) (factx->n_threads * 2)) {
n = factx->n_threads;
fa_o_store_args_t args = { factx, dst, o_tile_src, q_start, kv_head, ib3, n_rows_g, rows_per_t };
worker_callback_t store_fn = factx->is_dst_fp32 ? fa_o_store_thread_f32 : fa_o_store_thread_f16;
if (n > 1) {
- worker_pool_run_func(wp, store_fn, &args, n);
+ work_queue_run(wp, store_fn, &args, n);
} else {
store_fn(1, 0, &args);
}
return;
}
- struct htp_thread_trace * tr = factx->octx->ctx ? &factx->octx->ctx->trace[i] : NULL;
+ struct htp_thread_trace * tr = &factx->octx->ctx->trace[i];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_SFM, (uint16_t) (args->q_start * G + vec_start * 64));
// Per-thread row scratch: thread i uses bufs at offset i * 2 * stride
fa_softmax_args_t * sargs,
size_t n_row_tiles,
size_t n_row_tiles_g_br) {
- worker_pool_context_t wp = factx->octx->ctx->worker_pool;
+ work_queue_t wp = factx->octx->ctx->work_queue;
const size_t n_row_vec_cnt = hmx_ceil_div(sargs->n_rows_g, 64);
worker_callback_t softmax_fn = fa_softmax_thread;
if (factx->n_threads > 1 && n_row_vec_cnt >= 2) {
uint32_t n_use = (uint32_t) hex_smin((size_t) factx->n_threads, n_row_vec_cnt);
sargs->thread_div = init_fastdiv_values(n_use);
- worker_pool_run_func(wp, softmax_fn, sargs, n_use);
+ work_queue_run(wp, softmax_fn, sargs, n_use);
} else {
softmax_fn(1, 0, sargs);
}
// ============================================================================
int hmx_flash_attn_ext(struct htp_ops_context * octx) {
- struct htp_thread_trace * tr_hvx = octx->ctx ? &octx->ctx->trace[0] : NULL;
- struct htp_thread_trace * tr_hmx = octx->ctx ? &octx->ctx->trace[HTP_MAX_NTHREADS] : NULL;
+ struct htp_thread_trace * tr_hvx = &octx->ctx->trace[0];
+ struct htp_thread_trace * tr_hmx = &octx->ctx->trace[HTP_MAX_NTHREADS];
const struct htp_tensor * q = octx->src[0];
const struct htp_tensor * k = octx->src[1];
const struct htp_tensor * v = octx->src[2];
const size_t k_src_stride = size_k_row_padded / sizeof(__fp16);
const size_t v_src_stride = size_v_row_padded / sizeof(__fp16);
- struct hmx_queue * hmx_q = ctx->hmx_queue;
+ hmx_queue_t hmx_q = ctx->hmx_queue;
if (factx.pipeline) {
// Pipeline path
}
if (!(octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)) {
- worker_pool_run_func(octx->ctx->worker_pool, flash_attn_ext_f16_thread, &factx, octx->n_threads);
+ work_queue_run(octx->ctx->work_queue, flash_attn_ext_f16_thread, &factx, octx->n_threads);
}
return HTP_STATUS_OK;
--- /dev/null
+#ifndef HEX_BITMAP_H
+#define HEX_BITMAP_H
+
+#include <stdint.h>
+#include <stdbool.h>
+#include <string.h>
+
+static inline void bitmap_set(uint32_t * bitmap, uint32_t idx) {
+ bitmap[idx / 32] |= (1U << (idx % 32));
+}
+
+static inline void bitmap_clear(uint32_t * bitmap, uint32_t idx) {
+ bitmap[idx / 32] &= ~(1U << (idx % 32));
+}
+
+static inline bool bitmap_test(const uint32_t * bitmap, uint32_t idx) {
+ return (bitmap[idx / 32] & (1U << (idx % 32))) != 0;
+}
+
+static inline void bitmap_reset(uint32_t * bitmap, size_t size_in_bits) {
+ memset(bitmap, 0, ((size_in_bits + 31) / 32) * sizeof(uint32_t));
+}
+
+#endif // HEX_BITMAP_H
+++ /dev/null
-#include "hex-dma.h"
-
-#include <stdbool.h>
-#include <stdlib.h>
-#include <string.h>
-
-#pragma clang diagnostic ignored "-Wunused-function"
-
-static inline uint32_t pow2_ceil(uint32_t x) {
- if (x <= 1) {
- return 1;
- }
- int p = 2;
- x--;
- while (x >>= 1) {
- p <<= 1;
- }
- return p;
-}
-
-dma_queue * dma_queue_create(size_t capacity) {
- dma_queue * q = (dma_queue *) memalign(32, sizeof(dma_queue));
- if (q == NULL) {
- FARF(ERROR, "%s: failed to allocate DMA queue\n", __FUNCTION__);
- return NULL;
- }
-
- capacity = pow2_ceil(capacity);
-
- memset(q, 0, sizeof(dma_queue));
- q->capacity = capacity;
- q->idx_mask = capacity - 1;
-
- q->desc = (dma_descriptor_2d *) memalign(64, capacity * sizeof(dma_descriptor_2d));
- memset(q->desc, 0, capacity * sizeof(dma_descriptor_2d));
-
- q->dptr = (dma_ptr *) memalign(4, capacity * sizeof(dma_ptr));
- memset(q->dptr, 0, capacity * sizeof(dma_ptr));
-
- q->tail = &q->desc[capacity - 1];
-
- if (!q->desc && !q->dptr) {
- FARF(ERROR, "%s: failed to allocate DMA queue items\n", __FUNCTION__);
- return NULL;
- }
-
- FARF(HIGH, "dma-queue: capacity %u\n", capacity);
-
- return q;
-}
-
-void dma_queue_delete(dma_queue * q) {
- if (!q) {
- return;
- }
- free(q->desc);
- free(q->dptr);
- free(q);
-}
-
-void dma_queue_flush(dma_queue * q) {
- while (dma_queue_pop(q).dst != NULL) ;
-}
-#ifndef HTP_DMA_H
-#define HTP_DMA_H
-
-#include <HAP_farf.h>
-#include <hexagon_types.h>
-#include <stdbool.h>
-#include <stdint.h>
-#include "hex-utils.h"
-
-#include "hex-profile.h"
-
-#ifdef __cplusplus
-extern "C" {
-#endif
-
-// Define the HW descriptor structs here since the ones in HexSDK are a bit out of date
-typedef struct dma_descriptor_1d_s {
- void * next;
- uint32_t size:24;
- uint32_t desc_size:2;
- uint32_t dst_comp:1;
- uint32_t src_comp:1;
- uint32_t dst_bypass:1;
- uint32_t src_bypass:1;
- uint32_t order:1;
- uint32_t done:1;
- void * src;
- void * dst;
-} dma_descriptor_1d;
-
-#if __HVX_ARCH__ < 75
-
-typedef struct dma_descriptor_2d_s {
- void * next;
- uint32_t reserved0:24;
- uint32_t desc_size:2;
- uint32_t dst_comp:1;
- uint32_t src_comp:1;
- uint32_t dst_bypass:1;
- uint32_t src_bypass:1;
- uint32_t order:1;
- uint32_t done:1;
- void * src;
- void * dst;
- uint32_t desc_type:8;
- uint32_t reserved1:24;
- uint32_t row_size:16;
- uint32_t nrows:16;
- uint32_t src_stride:16;
- uint32_t dst_stride:16;
- uint32_t src_offset:16;
- uint32_t dst_offset:16;
-} dma_descriptor_2d;
-
-#else
-
-typedef struct dma_descriptor_2d_s {
- void * next;
- uint32_t dst_stride:24;
- uint32_t desc_size:2;
- uint32_t dst_comp:1;
- uint32_t src_comp:1;
- uint32_t dst_bypass:1;
- uint32_t src_bypass:1;
- uint32_t order:1;
- uint32_t done:1;
- void * src;
- void * dst;
- uint32_t desc_type:8;
- uint32_t reserved0:24;
- uint32_t row_size:24;
- uint32_t nrows_lo:8;
- uint32_t nrows_hi:8;
- uint32_t src_stride:24;
- uint32_t offset:24;
- uint32_t reserved1:8;
-} dma_descriptor_2d;
-
-#endif
-
-typedef struct {
- void *dst;
- const void *src;
-} dma_ptr;
-
-typedef struct {
- dma_descriptor_2d * desc; // descriptor pointers
- dma_descriptor_2d * tail; // tail pointer
- dma_ptr * dptr; // dst/src pointers
- uint32_t push_idx;
- uint32_t pop_idx;
- uint32_t capacity;
- uint32_t idx_mask;
- struct htp_thread_trace * trace;
-} dma_queue;
-
-dma_queue * dma_queue_create(size_t capacity);
-void dma_queue_delete(dma_queue * q);
-void dma_queue_flush(dma_queue * q);
-
-// TODO: technically we don't need these and could use Q6_dmstart/wait/etc instead
-// but those do not seem to always compiler properly.
-static inline void dmstart(void * next) {
- asm volatile(" release(%0):at" : : "r"(next));
- asm volatile(" dmstart(%0)" : : "r"(next));
-}
-
-static inline void dmlink(void * cur, void * next) {
- asm volatile(" release(%0):at" : : "r"(next));
- asm volatile(" dmlink(%0, %1)" : : "r"(cur), "r"(next));
-}
-
-static inline unsigned int dmpoll(void) {
- unsigned int ret = 0;
- asm volatile(" %0 = dmpoll" : "=r"(ret) : : "memory");
- return ret;
-}
-
-static inline unsigned int dmwait(void) {
- unsigned int ret = 0;
- asm volatile(" %0 = dmwait" : "=r"(ret) : : "memory");
- return ret;
-}
-
-static inline dma_ptr dma_make_ptr(void *dst, const void *src)
-{
- dma_ptr p = { dst, src };
- return p;
-}
-
-static const uint32_t dma_src_l2_bypass_on = 1;
-static const uint32_t dma_dst_l2_bypass_on = 1;
-
-static inline bool dma_queue_push_single_1d(dma_queue * q, dma_ptr dptr, size_t size) {
- if (((q->push_idx + 1) & q->idx_mask) == q->pop_idx) {
- FARF(HIGH, "dma-push: queue full\n");
- return false;
- }
-
- dma_descriptor_1d * desc = (dma_descriptor_1d *) &q->desc[q->push_idx];
- desc->src = (void *) dptr.src;
- desc->dst = (void *) dptr.dst;
- desc->size = size;
-
- q->dptr[q->push_idx] = dptr;
-
- if (size) {
- desc->next = NULL;
- desc->desc_size = 0; // 1D mode
- desc->src_bypass = dma_src_l2_bypass_on;
- desc->dst_bypass = dma_dst_l2_bypass_on;
- desc->order = 0;
- desc->done = 0;
-
- htp_trace_event_start(q->trace, HTP_TRACE_EVT_DMA, q->push_idx);
- dmlink(q->tail, desc);
- q->tail = (dma_descriptor_2d *) desc;
- } else {
- desc->desc_size = 0;
- desc->done = 1;
- }
-
- q->push_idx = (q->push_idx + 1) & q->idx_mask;
- return true;
-}
-
-static inline bool dma_queue_push_single_2d(dma_queue * q, dma_ptr dptr, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) {
- if (((q->push_idx + 1) & q->idx_mask) == q->pop_idx) {
- FARF(HIGH, "dma-push: queue full\n");
- return false;
- }
-
- dma_descriptor_2d * desc = &q->desc[q->push_idx];
-
- desc->next = NULL;
- desc->reserved0 = 0;
- desc->reserved1 = 0;
- desc->desc_size = 1; // 2d mode
- desc->src_bypass = dma_src_l2_bypass_on;
- desc->dst_bypass = dma_dst_l2_bypass_on;
- desc->src_comp = 0;
- desc->dst_comp = 0;
- desc->order = 0;
- desc->done = 0;
- desc->src_stride = src_stride;
- desc->dst_stride = dst_stride;
- desc->src = (void *) dptr.src;
- desc->dst = (void *) dptr.dst;
- desc->row_size = row_size;
-
-#if __HVX_ARCH__ < 75
- desc->desc_type = 0; // 2d (16-bit) mode
- desc->nrows = nrows;
- desc->src_offset = 0;
- desc->dst_offset = 0;
-#else
- desc->desc_type = 9; // 2d (24-bit) mode
- desc->nrows_lo = (nrows & 0xff);
- desc->nrows_hi = (nrows >> 8);
- desc->offset = 0;
-#endif
-
- q->dptr[q->push_idx] = dptr;
-
- if (nrows) {
- htp_trace_event_start(q->trace, HTP_TRACE_EVT_DMA, q->push_idx);
- dmlink(q->tail, desc);
- q->tail = desc;
- } else {
- desc->done = 1;
- }
-
- // FARF(ERROR, "dma-push: i %u row-size %u nrows %d dst %p src %p\n", q->push_idx, row_size, nrows, dptr.dst, dptr.src);
- q->push_idx = (q->push_idx + 1) & q->idx_mask;
- return true;
-}
-
-static inline dma_ptr dma_queue_pop(dma_queue * q) {
- dma_ptr dptr = { NULL };
-
- if (q->push_idx == q->pop_idx) {
- return dptr;
- }
-
- dma_descriptor_2d * desc = &q->desc[q->pop_idx];
-
- // Wait for desc to complete
- if (!desc->done) {
- while (!desc->done) {
- dmpoll();
- }
- }
- htp_trace_event_stop(q->trace, HTP_TRACE_EVT_DMA, q->pop_idx);
-
- dptr = q->dptr[q->pop_idx];
-
- // FARF(ERROR, "dma-pop: i %u dst %p src %p\n", q->pop_idx, dptr.dst, dptr.src);
- q->pop_idx = (q->pop_idx + 1) & q->idx_mask;
- return dptr;
-}
-
-static inline dma_ptr dma_queue_pop_nowait(dma_queue * q) {
- dma_ptr dptr = { NULL };
-
- if (q->push_idx == q->pop_idx) {
- return dptr;
- }
-
- dptr = q->dptr[q->pop_idx];
-
- // FARF(ERROR, "dma-pop-nowait: i %u dst %p src %p\n", q->pop_idx, dptr.dst, dptr.src);
- q->pop_idx = (q->pop_idx + 1) & q->idx_mask;
- return dptr;
-}
-
-static inline bool dma_queue_empty(dma_queue * q) {
- return q->push_idx == q->pop_idx;
-}
-
-static inline uint32_t dma_queue_depth(dma_queue * q) {
- return (q->push_idx - q->pop_idx) & q->idx_mask;
-}
-
-static inline uint32_t dma_queue_capacity(dma_queue * q) {
- return q->capacity;
-}
-
-#if __HVX_ARCH__ < 75
-
-// Overflow-safe DMA push: all 2d descriptor fields (row_size, nrows, src_stride, dst_stride) are 16-bit, max 65535.
-// This version transparently handles values that exceed the 16-bit limit and submits chained DMA transtions.
-
-#define DMA_MAX_FIELD_VAL 65535u
-
-static inline bool dma_queue_push(dma_queue *q, dma_ptr dptr, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) {
- // Fast path: everything fits in 16 bits
- if (nrows == 0 || __builtin_expect(
- row_size <= DMA_MAX_FIELD_VAL &&
- nrows <= DMA_MAX_FIELD_VAL &&
- src_stride <= DMA_MAX_FIELD_VAL &&
- dst_stride <= DMA_MAX_FIELD_VAL, 1)) {
- return dma_queue_push_single_2d(q, dptr, dst_stride, src_stride, row_size, nrows);
- }
-
- // Contiguous block
- // Use 1d DMA mode which supports sizes up to 24-bits (16MB)
- if (nrows == 1 || (row_size == src_stride && row_size == dst_stride)) {
- size_t total = row_size * nrows;
- return dma_queue_push_single_1d(q, dptr, total);
- }
-
- // Stride overflow — fall back to row-by-row.
- {
- const uint8_t *src = (const uint8_t *) dptr.src;
- uint8_t *dst = (uint8_t *) dptr.dst;
- for (size_t r = 0; r < nrows; ++r) {
- dma_ptr p = dma_make_ptr(dst + r * dst_stride, src + r * src_stride);
- if (!dma_queue_push_single_1d(q, p, row_size))
- return false;
- if (r + 1 < nrows)
- dma_queue_pop(q);
- }
- return true;
- }
-}
-
-#else // HVX_ARCH >= 75
-
-static inline bool dma_queue_push(dma_queue *q, dma_ptr dptr, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) {
- // On v75 and up we always use 2d 24-bit mode
- return dma_queue_push_single_2d(q, dptr, dst_stride, src_stride, row_size, nrows);
-}
-
-#endif
-
-static inline bool dma_queue_push_ddr_to_vtcm(dma_queue * q, dma_ptr dptr, size_t dst_row_size, size_t src_row_size, size_t nrows) {
- return dma_queue_push(q, dptr, dst_row_size, src_row_size, src_row_size, nrows);
-}
-
-static inline bool dma_queue_push_vtcm_to_ddr(dma_queue * q, dma_ptr dptr, size_t dst_row_size, size_t src_row_size, size_t nrows) {
- return dma_queue_push(q, dptr, dst_row_size, src_row_size, dst_row_size, nrows);
-}
-
-#define DMA_CACHE_MAX_SIZE 256U
-
-typedef struct {
- uint8_t *base;
- uint32_t line_size;
- uint32_t capacity;
- uint32_t src[DMA_CACHE_MAX_SIZE];
- uint16_t age[DMA_CACHE_MAX_SIZE];
-} dma_cache;
-
-static inline void dma_cache_init(dma_cache *c, uint8_t *base, uint32_t line_size, uint32_t capacity)
-{
- c->capacity = (capacity > DMA_CACHE_MAX_SIZE) ? DMA_CACHE_MAX_SIZE : capacity;
- c->base = base;
- c->line_size = line_size;
-
- for (unsigned i=0; i < c->capacity; i++) {
- c->src[i] = 0;
- c->age[i] = 0;
- }
-}
-
-static inline bool dma_cache_push(dma_queue *q, dma_cache *c, const uint8_t * src, uint32_t dst_stride, uint32_t src_stride, uint32_t row_size, uint32_t nrows)
-{
- uint32_t o_idx = 0;
- uint16_t o_age = 0;
- uint8_t * dst = 0;
-
- for (unsigned i=0; i < c->capacity; i++) {
- if (c->src[i] == (uint32_t) src) {
- c->age[i] = 0;
- dst = c->base + (i * c->line_size); nrows = 0; // dummy dma
- } else {
- c->age[i]++;
- if (c->age[i] > o_age) { o_age = c->age[i]; o_idx = i; }
- }
- }
- if (!dst) {
- c->age[o_idx] = 0;
- c->src[o_idx] = (uint32_t) src;
- dst = c->base + o_idx * c->line_size; // normal nrows dma
- return dma_queue_push(q, dma_make_ptr(dst, src), dst_stride, src_stride, row_size, nrows);
- }
-
- return dma_queue_push_single_1d(q, dma_make_ptr(dst, src), 0);
-}
-
-#ifdef __cplusplus
-} // extern "C"
-#endif
-
-#endif /* HTP_DMA_H */
+#pragma once
+#include "dma-queue.h"
};
static inline void htp_trace_event(struct htp_thread_trace * tr, uint16_t id, uint16_t info, uint32_t type) {
- if (tr && tr->events && tr->count < tr->max_events) {
- uint32_t idx = tr->count;
- tr->events[idx].id = id;
- tr->events[idx].info = info | (type == HTP_TRACE_EVT_STOP ? 0x8000 : 0);
- tr->events[idx].cycles = (uint32_t) hex_get_cycles();
+ if (tr->count < tr->max_events) {
+ uint32_t i = tr->count;
+ tr->events[i].id = id;
+ tr->events[i].info = info | (type == HTP_TRACE_EVT_STOP ? 0x8000 : 0);
+ tr->events[i].cycles = (uint32_t) hex_get_cycles();
tr->count++;
}
}
Q6_l2fetch_AP((void *) p, control);
}
-#define HEX_L2_LINE_SIZE 64
-#define HEX_L2_FLUSH_SIZE (128 * 1024)
+static inline void hex_l2fetch_block(const void * addr, size_t size) {
+ if (size == 0) return;
+ const uint32_t width = 16384; // 16KB rows
+ const uint32_t height = (size + width - 1) / width;
+ hex_l2fetch(addr, width, width, height);
+}
+
+#define HEX_L2_LINE_SIZE 128
+#define HEX_L2_BLOCK_SIZE (HEX_L2_LINE_SIZE * 4) // flush granularity (lines per loop iteration)
+#define HEX_L2_FLUSH_WQ_THRESHOLD (4 * 1024)
+#define HEX_L2_FLUSH_ALL_THRESHOLD (4 * 1024 * 1024)
static inline void hex_l2flush(void * addr, size_t size) {
- if (size > HEX_L2_FLUSH_SIZE) {
- qurt_mem_cache_clean((qurt_addr_t) 0, 0, QURT_MEM_CACHE_FLUSH_INVALIDATE_ALL, QURT_MEM_DCACHE);
- } else {
- const uint32_t s = (uint32_t) addr;
- const uint32_t e = s + size;
- for (uint32_t i = s; i < e; i += HEX_L2_LINE_SIZE * 4) {
- Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 0);
- Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 1);
- Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 2);
- Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 3);
- }
+ const uint32_t s = ((uint32_t) addr) & ~(HEX_L2_LINE_SIZE - 1);
+ const uint32_t e = (((uint32_t) addr) + size + HEX_L2_LINE_SIZE - 1) & ~(HEX_L2_LINE_SIZE - 1);
+ for (uint32_t i = s; i < e; i += HEX_L2_BLOCK_SIZE) {
+ Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 0);
+ Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 1);
+ Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 2);
+ Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 3);
}
}
}
}
+static void transfer_activation_row_pair_fp32_to_fp16_col_chunk(
+ __fp16 *restrict vtcm_dst,
+ const float *restrict row0, // offset by c_first
+ const float *restrict row1, // offset by c_first
+ uint32_t r,
+ uint32_t k_block,
+ uint32_t c_first,
+ uint32_t c_len,
+ uint32_t k_chunk_valid,
+ bool row0_valid,
+ bool row1_valid) {
+
+ uint32_t r0 = r / HTP_MM_HMX_TILE_N_ROWS; // tile row index
+ uint32_t r1 = r % HTP_MM_HMX_TILE_N_ROWS; // intra-tile row idx
+
+ uint32_t c = 0;
+ for (; c + 32 <= k_chunk_valid; c += 32) {
+ HVX_Vector v0 = Q6_V_vzero();
+ HVX_Vector v1 = Q6_V_vzero();
+ if (row0_valid) v0 = *(const HVX_Vector *)(row0 + c);
+ if (row1_valid) v1 = *(const HVX_Vector *)(row1 + c);
+
+ HVX_Vector v_out = hvx_vec_f32_to_f16_shuff(v0, v1);
+
+ uint32_t c0 = (c_first + c) / HTP_MM_HMX_TILE_N_COLS; // tile column index
+ uint32_t tile_idx = r0 * (k_block / HTP_MM_HMX_TILE_N_COLS) + c0;
+
+ HVX_Vector *tile = (HVX_Vector *) (vtcm_dst + tile_idx * HTP_MM_HMX_TILE_N_ELMS);
+ tile[r1 / 2] = v_out;
+ }
+ if (c < c_len) {
+ HVX_Vector v0 = Q6_V_vzero();
+ HVX_Vector v1 = Q6_V_vzero();
+ if (row0_valid) v0 = *(const HVX_Vector *)(row0 + c);
+ if (row1_valid) v1 = *(const HVX_Vector *)(row1 + c);
+
+ uint32_t rem = (k_chunk_valid > c) ? (k_chunk_valid - c) : 0;
+ HVX_VectorPred mask = Q6_Q_vsetq2_R(rem > 0 ? rem * sizeof(float) : 0);
+ v0 = Q6_V_vmux_QVV(mask, v0, Q6_V_vzero());
+ v1 = Q6_V_vmux_QVV(mask, v1, Q6_V_vzero());
+
+ HVX_Vector v_out = hvx_vec_f32_to_f16_shuff(v0, v1);
+
+ uint32_t c0 = (c_first + c) / HTP_MM_HMX_TILE_N_COLS; // tile column index
+ uint32_t tile_idx = r0 * (k_block / HTP_MM_HMX_TILE_N_COLS) + c0;
+
+ HVX_Vector *tile = (HVX_Vector *) (vtcm_dst + tile_idx * HTP_MM_HMX_TILE_N_ELMS);
+ tile[r1 / 2] = v_out;
+ }
+}
+
static void transfer_activation_chunk_fp32_to_fp16_gathered(
__fp16 *restrict vtcm_dst,
const float *restrict src,
uint32_t start_row,
+ uint32_t vtcm_start_row,
uint32_t n_rows,
uint32_t k_block,
const struct mmid_row_mapping *matrix_rows,
for (r = 0; r < n_rows_tiled; r += 2) {
uint32_t r_idx0 = start_row + r + 0;
uint32_t r_idx1 = start_row + r + 1;
- uint32_t r0 = r_idx0 / HTP_MM_HMX_TILE_N_ROWS; // tile row index
- uint32_t r1 = r_idx0 % HTP_MM_HMX_TILE_N_ROWS; // intra-tile row idx
+ uint32_t lr = vtcm_start_row + r; // vtcm-local row
+ uint32_t r0 = lr / HTP_MM_HMX_TILE_N_ROWS; // tile row index
+ uint32_t r1 = lr % HTP_MM_HMX_TILE_N_ROWS; // intra-tile row idx
struct mmid_row_mapping mapping0 = matrix_rows[cur_a * mapping_stride + r_idx0];
struct mmid_row_mapping mapping1 = matrix_rows[cur_a * mapping_stride + r_idx1];
}
for (; r < n_rows_padded; r += 2) {
- uint32_t r_idx0 = start_row + r;
- uint32_t r0 = r_idx0 / HTP_MM_HMX_TILE_N_ROWS; // tile row index
- uint32_t r1 = r_idx0 % HTP_MM_HMX_TILE_N_ROWS; // intra-tile row idx
+ uint32_t lr = vtcm_start_row + r; // vtcm-local row
+ uint32_t r0 = lr / HTP_MM_HMX_TILE_N_ROWS; // tile row index
+ uint32_t r1 = lr % HTP_MM_HMX_TILE_N_ROWS; // intra-tile row idx
const bool row0_valid = (start_row + r + 0) < cne1;
const bool row1_valid = (start_row + r + 1) < cne1;
__fp16 *restrict vtcm_dst,
const float *restrict src,
uint32_t start_row,
+ uint32_t vtcm_start_row,
uint32_t n_rows,
uint32_t k_block,
const struct mmid_row_mapping *matrix_rows,
for (r = 0; r < n_rows_tiled; r += 2) {
uint32_t r_idx0 = start_row + r + 0;
uint32_t r_idx1 = start_row + r + 1;
- uint32_t r0 = r_idx0 / HTP_MM_HMX_TILE_N_ROWS; // tile row index
- uint32_t r1 = r_idx0 % HTP_MM_HMX_TILE_N_ROWS; // intra-tile row idx
+ uint32_t lr = vtcm_start_row + r; // vtcm-local row
+ uint32_t r0 = lr / HTP_MM_HMX_TILE_N_ROWS; // tile row index
+ uint32_t r1 = lr % HTP_MM_HMX_TILE_N_ROWS; // intra-tile row idx
struct mmid_row_mapping mapping0 = matrix_rows[cur_a * mapping_stride + r_idx0];
struct mmid_row_mapping mapping1 = matrix_rows[cur_a * mapping_stride + r_idx1];
}
for (; r < n_rows_padded; r += 2) {
- uint32_t r_idx0 = start_row + r;
- uint32_t r0 = r_idx0 / HTP_MM_HMX_TILE_N_ROWS; // tile row index
- uint32_t r1 = r_idx0 % HTP_MM_HMX_TILE_N_ROWS; // intra-tile row idx
+ uint32_t lr = vtcm_start_row + r; // vtcm-local row
+ uint32_t r0 = lr / HTP_MM_HMX_TILE_N_ROWS; // tile row index
+ uint32_t r1 = lr % HTP_MM_HMX_TILE_N_ROWS; // intra-tile row idx
const bool row0_valid = (start_row + r + 0) < cne1;
const bool row1_valid = (start_row + r + 1) < cne1;
float *restrict dst,
const __fp16 *restrict vtcm_src,
uint32_t start_row,
+ uint32_t vtcm_start_row,
uint32_t n_rows,
uint32_t n_cols,
const struct mmid_row_mapping *matrix_rows,
for (size_t r = 0; r < n_rows; r += 2) {
uint32_t r_idx0 = start_row + r + 0;
uint32_t r_idx1 = start_row + r + 1;
- const size_t r0 = r_idx0 / HTP_MM_HMX_TILE_N_ROWS;
- const size_t r1 = (r_idx0 % HTP_MM_HMX_TILE_N_ROWS) / 2; // index of the row pair within the tile
+ uint32_t lr = vtcm_start_row + r; // vtcm-local row
+ const size_t r0 = (lr / HTP_MM_HMX_TILE_N_ROWS);
+ const size_t r1 = (lr % HTP_MM_HMX_TILE_N_ROWS) / 2; // index of the row pair within the tile
const __fp16 *row_base = vtcm_src + r0 * tile_row_stride;
if (r_idx0 >= cne1) break;
#define QURT_LOWEST_PRIO (254)
-static inline void hmx_lock(struct hmx_queue *q)
+static inline void hmx_lock(hmx_queue_t q)
{
if (!q->hmx_locked) {
HAP_compute_res_hmx_lock(q->hap_rctx);
}
}
-static inline void hmx_unlock(struct hmx_queue *q)
+static inline void hmx_unlock(hmx_queue_t q)
{
if (q->hmx_locked) {
HAP_compute_res_hmx_unlock(q->hap_rctx);
}
}
-static inline void hmx_queue_process(struct hmx_queue *q, bool* killed) {
+static inline void hmx_queue_process(hmx_queue_t q, bool* killed) {
unsigned int ir = atomic_load(&q->idx_read);
while (ir != atomic_load(&q->idx_write)) {
}
static void hmx_queue_thread(void * arg) {
- struct hmx_queue * q = (struct hmx_queue *) arg;
+ hmx_queue_t q = (hmx_queue_t) arg;
FARF(HIGH, "hmx-queue-thread: started");
FARF(HIGH, "hmx-queue-thread: stopped");
}
-struct hmx_queue * hmx_queue_create(size_t capacity, uint32_t hap_rctx) {
+size_t hmx_queue_sizeof(size_t capacity, uint32_t stack_size) {
capacity = hex_ceil_pow2(capacity);
+ size_t size_q = hex_align_up(sizeof(struct hmx_queue_s), HEX_L2_LINE_SIZE);
+ size_t size_desc = hex_align_up(capacity * sizeof(struct hmx_queue_desc), HEX_L2_LINE_SIZE);
+ size_t size_stack = stack_size;
+ return size_q + size_desc + size_stack;
+}
+
+size_t hmx_queue_alignof(void) {
+ return HEX_L2_LINE_SIZE;
+}
+
+hmx_queue_t hmx_queue_init(void * ptr, size_t capacity, uint32_t stack_size, uint32_t hap_rctx, struct htp_thread_trace * trace) {
+ capacity = hex_ceil_pow2(capacity);
+ size_t size_q = hex_align_up(sizeof(struct hmx_queue_s), HEX_L2_LINE_SIZE);
+ size_t size_desc = hex_align_up(capacity * sizeof(struct hmx_queue_desc), HEX_L2_LINE_SIZE);
+
+ uint8_t * block = (uint8_t *) ptr;
+
+ hmx_queue_t q = (hmx_queue_t) block; block += size_q;
+ memset(q, 0, sizeof(struct hmx_queue_s));
- struct hmx_queue * q = (struct hmx_queue *) memalign(32, sizeof(struct hmx_queue));
- if (q == NULL) {
- FARF(ERROR, "%s: failed to allocate DMA queue\n", __FUNCTION__);
- return NULL;
- }
- memset(q, 0, sizeof(struct hmx_queue));
q->capacity = capacity;
q->idx_mask = capacity - 1;
q->hap_rctx = hap_rctx;
+ q->external_mem = true;
- q->desc = (struct hmx_queue_desc *) memalign(64, capacity * sizeof(struct hmx_queue_desc));
- if (!q->desc) {
- FARF(ERROR, "hmx-queue: failed to allocate HMX queue descriptors\n");
- return NULL;
- }
+ q->desc = (struct hmx_queue_desc *) block; block += size_desc;
memset(q->desc, 0, capacity * sizeof(struct hmx_queue_desc));
- const size_t stack_size = HMX_QUEUE_THREAD_STACK_SIZE;
- q->stack = (unsigned char *) memalign(64, stack_size);
- if (!q->stack) {
- FARF(ERROR, "hmx-queue: thread stack allocation failed (%zu bytes)", stack_size);
- return NULL;
- }
+ q->stack = block;
memset(q->stack, 0, stack_size);
+ q->trace = trace;
+
// Match caller thread priority (same pattern as worker-pool.c).
int prio = qurt_thread_get_priority(qurt_thread_get_id());
if (prio < 1) {
return q;
}
-void hmx_queue_delete(struct hmx_queue * q) {
+void hmx_queue_free(hmx_queue_t q) {
if (!q) {
return;
}
int status;
qurt_thread_join(q->thread, &status);
-
- free(q->desc);
- free(q->stack);
- free(q);
}
extern "C" {
#endif
-#define HMX_QUEUE_THREAD_STACK_SIZE (16 * 1024)
-
#if __HVX_ARCH__ > 79
#define HMX_QUEUE_POLL_COUNT 2000
#else
atomic_uint done;
};
-struct hmx_queue {
+struct hmx_queue_s {
struct hmx_queue_desc * desc;
atomic_uint idx_write; // updated by producer (push)
atomic_uint idx_read; // updated by consumer (process)
uint32_t hap_rctx;
bool hmx_locked;
struct htp_thread_trace * trace;
+ bool external_mem; // memory owned externally
};
-struct hmx_queue * hmx_queue_create(size_t capacity, uint32_t hap_rctx);
-void hmx_queue_delete(struct hmx_queue * q);
+typedef struct hmx_queue_s * hmx_queue_t;
+
+size_t hmx_queue_sizeof(size_t capacity, uint32_t stack_size);
+size_t hmx_queue_alignof(void);
+hmx_queue_t hmx_queue_init(void * ptr, size_t capacity, uint32_t stack_size, uint32_t hap_rctx, struct htp_thread_trace * trace);
+void hmx_queue_free(hmx_queue_t q);
static inline struct hmx_queue_desc hmx_queue_make_desc(hmx_queue_func func, void * data) {
struct hmx_queue_desc d = { func, data };
return d;
}
-static inline bool hmx_queue_push(struct hmx_queue * q, struct hmx_queue_desc d) {
+static inline bool hmx_queue_push(hmx_queue_t q, struct hmx_queue_desc d) {
unsigned int ir = atomic_load(&q->idx_read);
- unsigned int iw = q->idx_write;
+ unsigned int iw = atomic_load(&q->idx_write);
if (((iw + 1) & q->idx_mask) == ir) {
FARF(HIGH, "hmx-queue-push: queue is full\n");
return true;
}
-static inline bool hmx_queue_signal(struct hmx_queue *q, enum hmx_queue_signal sig) {
+static inline bool hmx_queue_signal(hmx_queue_t q, enum hmx_queue_signal sig) {
return hmx_queue_push(q, hmx_queue_make_desc((hmx_queue_func) sig, NULL));
}
-static inline bool hmx_queue_empty(struct hmx_queue * q) {
- return q->idx_pop == q->idx_write;
+static inline bool hmx_queue_empty(hmx_queue_t q) {
+ return q->idx_pop == atomic_load(&q->idx_write);
}
-static inline uint32_t hmx_queue_depth(struct hmx_queue * q) {
- return (q->idx_read - q->idx_read) & q->idx_mask;
+static inline uint32_t hmx_queue_depth(hmx_queue_t q) {
+ return (atomic_load(&q->idx_write) - atomic_load(&q->idx_read)) & q->idx_mask;
}
-static inline uint32_t hmx_queue_capacity(struct hmx_queue * q) {
+static inline uint32_t hmx_queue_capacity(hmx_queue_t q) {
return q->capacity;
}
-static inline struct hmx_queue_desc hmx_queue_pop_one(struct hmx_queue * q) {
+static inline struct hmx_queue_desc hmx_queue_pop_one(hmx_queue_t q) {
unsigned int ip = q->idx_pop;
- unsigned int iw = q->idx_write;
+ unsigned int iw = atomic_load(&q->idx_write);
struct hmx_queue_desc rd = { NULL, NULL };
if (ip == iw) {
return rd;
}
-static inline struct hmx_queue_desc hmx_queue_pop(struct hmx_queue * q) {
+static inline struct hmx_queue_desc hmx_queue_pop(hmx_queue_t q) {
while (1) {
struct hmx_queue_desc d = hmx_queue_pop_one(q);
}
}
-static inline void hmx_queue_flush(struct hmx_queue * q) {
+static inline void hmx_queue_flush(hmx_queue_t q) {
while (hmx_queue_pop_one(q).func != NULL) ;
}
-static inline void hmx_queue_wakeup(struct hmx_queue * q) {
+static inline void hmx_queue_wakeup(hmx_queue_t q) {
hmx_queue_signal(q, HMX_QUEUE_WAKEUP);
}
-static inline void hmx_queue_suspend(struct hmx_queue *q) {
+static inline void hmx_queue_suspend(hmx_queue_t q) {
hmx_queue_signal(q, HMX_QUEUE_SUSPEND);
}
#include "hmx-queue.h"
#include "htp-ops.h"
#include "hex-profile.h"
-#include "worker-pool.h"
+#include "work-queue.h"
+#include "hex-fastdiv.h"
#include <assert.h>
#include <dspqueue.h>
const struct htp_tensor * dsts[HTP_OP_MAX_OUTPUTS];
};
+ dma_queue ** src_dma[HTP_OP_MAX_INPUTS];
+ dma_queue ** dst_dma[HTP_OP_MAX_OUTPUTS];
+
// TODO convert these to an array
struct htp_spad src0_spad;
struct htp_spad src1_spad;
// Main context for htp DSP backend
struct htp_context {
- dspqueue_t queue;
- dma_queue * dma[HTP_MAX_NTHREADS];
+ dspqueue_t dsp_queue;
+
struct htp_mmap mmap[HTP_MAX_MMAPS];
- worker_pool_context_t worker_pool;
+ dma_queue_t dma[HTP_MAX_NTHREADS];
+ dma_queue_t dma_cached[HTP_MAX_NTHREADS];
+ work_queue_t work_queue;
+ hmx_queue_t hmx_queue;
+
uint32_t n_threads;
+ struct fastdiv_values n_threads_div;
int thread_id;
int thread_prio;
atomic_bool vtcm_needs_release;
uint64_t max_vmem;
+ uint32_t dirty_map[HTP_OP_MAX_TENSORS / 32];
// Persistent DDR scratchpad for MUL_MAT_ID mappings
void * ddr_spad_base;
struct htp_ops_context octx;
- struct hmx_queue * hmx_queue; // Async HMX queue for pipeline overlap
+ qurt_thread_t main_thread;
+ void * main_stack;
+ atomic_bool killed;
+ size_t footprint;
};
int op_matmul(struct htp_ops_context * octx);
#define HTP_OP_MAX_KERN_PARAMS 32
#define HTP_OP_MAX_BUFS 16
-#define HTP_OP_MAX_REQS 256
-#define HTP_OP_MAX_TENSORS (HTP_OP_MAX_REQS * HTP_OP_MAX_INPUTS + HTP_OP_MAX_REQS)
+#define HTP_OP_MAX_TENSORS 8192 // must stay under 64K (uint16)
#define HTP_OP_MAX_VMEM_DEFAULT (3355443200u)
enum htp_tensor_flags {
HTP_TENSOR_COMPUTE = (1U << 0), // Tensor buffer temporal compute data (not weights)
- HTP_TENSOR_FLUSHED = (1U << 1) // Tensor buffer has been flushed (set by the NPU)
+ HTP_TENSOR_DIRTY = (1U << 1) // Tensor buffer is dirty and needs to be flushed
};
// Tensor descriptor
struct htp_tensor {
uint32_t data; // Buffer offset in the messages, and data pointer on the NPU
+ uint32_t alias; // Index of the canonical tensor for this memory buffer
uint32_t size; // Data size in bytes
uint32_t flags; // Buffer / tensor flags
- uint16_t type; // Data type
+ uint32_t type; // Data type
uint16_t bi; // Buffer index
+ uint16_t ti; // Tensor index
uint32_t ne[HTP_OP_MAX_DIMS]; // Number of elements
uint32_t nb[HTP_OP_MAX_DIMS]; // Stride in bytes (see ggml.h ggml_tensor)
};
enum htp_trace_event_id {
HTP_TRACE_EVT_DMA = 0,
+ HTP_TRACE_EVT_L2FLUSH = 1,
+ HTP_TRACE_EVT_INIT = 2,
HTP_TRACE_EVT_HVX_COMP = 20,
HTP_TRACE_EVT_HVX_A_QUANT = 21,
--- /dev/null
+#include "htp-tensor.h"
+
+#include <qurt.h>
+#include <qurt_memory.h>
+
+#include "hex-common.h"
+#include "hex-utils.h"
+#include "hex-fastdiv.h"
+#include "hex-profile.h"
+#include "htp-ctx.h"
+#include "work-queue.h"
+
+struct l2flush_task {
+ struct htp_thread_trace * trace;
+ uint32_t start;
+ uint32_t end;
+ uint32_t chunk_size;
+ uint32_t ti;
+};
+
+static void l2flush_thread_worker(unsigned int n, unsigned int i, void * data) {
+ struct l2flush_task * task = (struct l2flush_task *) data;
+ const uint32_t start = task->start;
+ const uint32_t end = task->end;
+ const uint32_t ti = task->ti;
+ const uint32_t chunk_size = task->chunk_size;
+
+ const uint32_t thread_s = start + i * chunk_size;
+ if (thread_s >= end) {
+ return;
+ }
+ uint32_t thread_e = thread_s + chunk_size;
+ if (thread_e > end) {
+ thread_e = end;
+ }
+
+ struct htp_thread_trace * tr = &task->trace[i];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_L2FLUSH, ti);
+ hex_l2flush((void *) (uintptr_t) thread_s, thread_e - thread_s);
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_L2FLUSH, ti);
+}
+
+static void flush_all_dcache(struct htp_context * ctx) {
+ struct htp_thread_trace * tr = &ctx->trace[0];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_L2FLUSH, 0);
+ qurt_mem_cache_clean((qurt_addr_t) 0, 0, QURT_MEM_CACHE_FLUSH_INVALIDATE_ALL, QURT_MEM_DCACHE);
+ hex_l2fetch_block(ctx, ctx->footprint);
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_L2FLUSH, 0);
+ bitmap_reset(ctx->dirty_map, HTP_OP_MAX_TENSORS);
+}
+
+static void flush_tensor_range(struct htp_context * ctx, const struct htp_tensor * t) {
+ struct htp_thread_trace * tr = &ctx->trace[0];
+
+ if (t->size > HEX_L2_FLUSH_WQ_THRESHOLD && ctx->n_threads > 1) {
+ struct l2flush_task task;
+ task.start = hex_align_down((size_t) t->data, HEX_L2_LINE_SIZE);
+ task.end = hex_align_up((size_t) t->data + t->size, HEX_L2_LINE_SIZE);
+ task.ti = t->ti;
+ task.trace = ctx->trace;
+
+ const uint32_t total_size = task.end - task.start;
+ const uint32_t n_blocks = (total_size + HEX_L2_BLOCK_SIZE - 1) / HEX_L2_BLOCK_SIZE;
+ const uint32_t blocks_per_thread = fastdiv(n_blocks + ctx->n_threads - 1, &ctx->n_threads_div);
+ task.chunk_size = blocks_per_thread * HEX_L2_BLOCK_SIZE;
+
+ work_queue_run(ctx->work_queue, l2flush_thread_worker, &task, ctx->n_threads);
+ } else {
+ htp_trace_event_start(tr, HTP_TRACE_EVT_L2FLUSH, t->ti);
+ hex_l2flush((void *) t->data, t->size);
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_L2FLUSH, t->ti);
+ }
+
+ htp_tensor_make_clean(t, ctx->dirty_map);
+}
+
+void htp_tensor_flush(struct htp_context * ctx, const struct htp_tensor * t) {
+ if (!bitmap_test(ctx->dirty_map, t->ti)) {
+ return;
+ }
+
+ if (t->size > HEX_L2_FLUSH_ALL_THRESHOLD) {
+ flush_all_dcache(ctx);
+ return;
+ }
+
+ flush_tensor_range(ctx, t);
+}
+
+// One dirty tensor's line-aligned range, placed in the flattened global block space.
+struct l2flush_range {
+ uint32_t start; // line-aligned start address
+ uint32_t end; // line-aligned end address
+ uint32_t block_first; // global block index of this range's first block
+ uint32_t n_blocks; // number of HEX_L2_BLOCK_SIZE chunks (last may be partial)
+};
+
+struct l2flush_multi_task {
+ struct htp_thread_trace * trace;
+ struct l2flush_range ranges[HTP_OP_MAX_INPUTS];
+ uint32_t n_ranges;
+ uint32_t total_blocks;
+ uint32_t blocks_per_thread;
+};
+
+static void l2flush_multi_worker(unsigned int n, unsigned int i, void * data) {
+ (void) n;
+ struct l2flush_multi_task * task = (struct l2flush_multi_task *) data;
+
+ const uint32_t gb_first = i * task->blocks_per_thread;
+ uint32_t gb_last = gb_first + task->blocks_per_thread;
+ if (gb_last > task->total_blocks) {
+ gb_last = task->total_blocks;
+ }
+ if (gb_first >= gb_last) {
+ return;
+ }
+
+ struct htp_thread_trace * tr = &task->trace[i];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_L2FLUSH, gb_first);
+
+ for (uint32_t r = 0; r < task->n_ranges; r++) {
+ const struct l2flush_range * rg = &task->ranges[r];
+ const uint32_t rb_first = rg->block_first;
+ const uint32_t rb_last = rg->block_first + rg->n_blocks;
+
+ const uint32_t lo = gb_first > rb_first ? gb_first : rb_first;
+ const uint32_t hi = gb_last < rb_last ? gb_last : rb_last;
+ if (lo >= hi) {
+ continue;
+ }
+
+ const uint32_t s = rg->start + (lo - rb_first) * HEX_L2_BLOCK_SIZE;
+ uint32_t e = rg->start + (hi - rb_first) * HEX_L2_BLOCK_SIZE;
+ if (e > rg->end) {
+ e = rg->end;
+ }
+ hex_l2flush((void *) (uintptr_t) s, e - s);
+ }
+
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_L2FLUSH, gb_first);
+}
+
+void htp_tensor_flush_all(struct htp_context * ctx, const struct htp_tensor * const * tensors, uint32_t n) {
+ uint64_t total_dirty = 0;
+ for (uint32_t i = 0; i < n; i++) {
+ const struct htp_tensor * t = tensors[i];
+ if (t && bitmap_test(ctx->dirty_map, t->ti)) {
+ total_dirty += t->size;
+ }
+ }
+
+ if (total_dirty == 0) {
+ return;
+ }
+
+ if (total_dirty > HEX_L2_FLUSH_ALL_THRESHOLD) {
+ flush_all_dcache(ctx);
+ return;
+ }
+
+ // Aggregate is small enough to walk. Thread it across all dirty ranges at once
+ // when it is worth the dispatch, otherwise flush sequentially.
+ if (total_dirty > HEX_L2_FLUSH_WQ_THRESHOLD && ctx->n_threads > 1) {
+ struct l2flush_multi_task task;
+ task.trace = ctx->trace;
+ task.n_ranges = 0;
+
+ uint32_t block_acc = 0;
+ for (uint32_t i = 0; i < n; i++) {
+ const struct htp_tensor * t = tensors[i];
+ if (!t || !bitmap_test(ctx->dirty_map, t->ti)) {
+ continue;
+ }
+ // Clear as we go: dedups a tensor passed as multiple srcs (e.g. mul(x,x)).
+ htp_tensor_make_clean(t, ctx->dirty_map);
+
+ struct l2flush_range * rg = &task.ranges[task.n_ranges++];
+ rg->start = hex_align_down((size_t) t->data, HEX_L2_LINE_SIZE);
+ rg->end = hex_align_up((size_t) t->data + t->size, HEX_L2_LINE_SIZE);
+ rg->block_first = block_acc;
+ rg->n_blocks = (rg->end - rg->start + HEX_L2_BLOCK_SIZE - 1) / HEX_L2_BLOCK_SIZE;
+ block_acc += rg->n_blocks;
+ }
+
+ task.total_blocks = block_acc;
+ task.blocks_per_thread = fastdiv(block_acc + ctx->n_threads - 1, &ctx->n_threads_div);
+
+ work_queue_run(ctx->work_queue, l2flush_multi_worker, &task, ctx->n_threads);
+ return;
+ }
+
+ struct htp_thread_trace * tr = &ctx->trace[0];
+ for (uint32_t i = 0; i < n; i++) {
+ const struct htp_tensor * t = tensors[i];
+ if (!t || !bitmap_test(ctx->dirty_map, t->ti)) {
+ continue;
+ }
+ htp_trace_event_start(tr, HTP_TRACE_EVT_L2FLUSH, t->ti);
+ hex_l2flush((void *) t->data, t->size);
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_L2FLUSH, t->ti);
+ htp_tensor_make_clean(t, ctx->dirty_map);
+ }
+}
--- /dev/null
+#ifndef HTP_TENSOR_H
+#define HTP_TENSOR_H
+
+#include <stdint.h>
+#include "htp-ops.h"
+#include "hex-bitmap.h"
+
+static inline struct htp_tensor * htp_tensor_alias(const struct htp_tensor * t) {
+ return (struct htp_tensor *) (uintptr_t) t->alias;
+}
+
+static inline void * htp_tensor_data(const struct htp_tensor * t) {
+ return (void *) (uintptr_t) t->data;
+}
+
+static inline uint32_t * htp_tensor_flags(const struct htp_tensor * t) {
+ return (uint32_t *) &t->flags;
+}
+
+static inline void htp_tensor_make_dirty(const struct htp_tensor * t, uint32_t * dirty_map) {
+ struct htp_tensor * curr = (struct htp_tensor *) t;
+ do {
+ bitmap_set(dirty_map, curr->ti);
+ curr = htp_tensor_alias(curr);
+ } while (curr != t);
+}
+
+static inline void htp_tensor_make_clean(const struct htp_tensor * t, uint32_t * dirty_map) {
+ bitmap_clear(dirty_map, t->ti);
+}
+
+struct htp_context;
+void htp_tensor_flush(struct htp_context * ctx, const struct htp_tensor * t);
+void htp_tensor_flush_all(struct htp_context * ctx, const struct htp_tensor * const * tensors, uint32_t n);
+
+#endif // HTP_TENSOR_H
#define GGML_COMMON_DECL_C
#include "ggml-common.h"
+#include "hex-bitmap.h"
#include "htp-ctx.h"
#include "htp-ops.h"
-#include "htp-ops.h"
+#include "htp-tensor.h"
#include "htp_iface.h"
-#include "worker-pool.h"
-
-AEEResult htp_iface_open(const char * uri, remote_handle64 * handle) {
- struct htp_context * ctx;
- int err = 0;
-
- ctx = calloc(1, sizeof(*ctx));
- if (ctx == NULL) {
- return AEE_ENOMEMORY;
- }
-
- // Use the context structure as the handle
- *handle = (remote_handle64) ctx;
-
- // Enable FARF logs
- HAP_setFARFRuntimeLoggingParams(0xffff, NULL, 0);
-
- // Set client class
- {
- HAP_power_request_t request;
- memset(&request, 0, sizeof(HAP_power_request_t));
- request.type = HAP_power_set_apptype;
- request.apptype = HAP_POWER_COMPUTE_CLIENT_CLASS;
-
- if ((err = HAP_power_set((void *) ctx, &request)) != 0) {
- return err;
- }
- }
-
- {
- HAP_power_request_t request;
- memset(&request, 0, sizeof(request));
+#include "work-queue.h"
+#include "hex-profile.h"
- request.type = HAP_power_set_DCVS_v3;
- request.dcvs_v3.set_dcvs_enable = TRUE;
- request.dcvs_v3.dcvs_enable = FALSE;
- request.dcvs_v3.set_bus_params = TRUE;
- request.dcvs_v3.bus_params.min_corner = HAP_DCVS_VCORNER_MAX;
- request.dcvs_v3.bus_params.max_corner = HAP_DCVS_VCORNER_MAX;
- request.dcvs_v3.bus_params.target_corner = HAP_DCVS_VCORNER_MAX;
- request.dcvs_v3.set_core_params = TRUE;
- request.dcvs_v3.core_params.min_corner = HAP_DCVS_VCORNER_MAX;
- request.dcvs_v3.core_params.max_corner = HAP_DCVS_VCORNER_MAX;
- request.dcvs_v3.core_params.target_corner = HAP_DCVS_VCORNER_MAX;
- request.dcvs_v3.set_sleep_disable = TRUE;
- request.dcvs_v3.sleep_disable = TRUE;
+#define HMX_QUEUE_CAPACITY 16
+#define HMX_QUEUE_STACK_SIZE 16384
+#define WORK_QUEUE_CAPACITY 16
+#define WORK_QUEUE_STACK_SIZE 16384
+#define MAIN_THREAD_STACK_SIZE 32768
-#if (__HEXAGON_ARCH__ >= 79)
- HAP_set_dcvs_v3_protected_bus_corners(&request, 1);
-#endif
- if ((err = HAP_power_set((void *) ctx, &request)) != 0) {
- return err;
- }
+_Static_assert(WORK_QUEUE_MAX_N_THREADS >= HTP_MAX_NTHREADS,
+ "work-queue thread cap must be >= HTP_MAX_NTHREADS");
- memset(&request, 0, sizeof(request));
- request.type = HAP_power_set_HVX;
- request.hvx.power_up = TRUE;
- if ((err = HAP_power_set((void *) ctx, &request)) != 0) {
- return err;
- }
- }
+struct htp_handle {
+ struct htp_context * ctx;
+};
-#if __HVX_ARCH__ >= 75
- {
- // Power on HMX and set HMX clock
- HAP_power_request_t request;
- memset(&request, 0, sizeof(HAP_power_request_t));
- request.type = HAP_power_set_HMX_v2;
- request.hmx_v2.set_power = TRUE;
- request.hmx_v2.power_up = TRUE;
- request.hmx_v2.set_clock = TRUE;
- request.hmx_v2.target_corner = HAP_DCVS_EXP_VCORNER_MAX;
- request.hmx_v2.min_corner = HAP_DCVS_EXP_VCORNER_MAX;
- request.hmx_v2.max_corner = HAP_DCVS_EXP_VCORNER_MAX;
- request.hmx_v2.perf_mode = HAP_CLK_PERF_HIGH;
- FARF(ALWAYS, "Setting HMX clock\n");
- err = HAP_power_set((void *) ctx, &request);
- if (err != AEE_SUCCESS) {
- FARF(ERROR, "ggml-hex: error setting HMX clock.");
- return err;
- }
- }
-#else
- {
- // Power on HMX
- HAP_power_request_t request;
- memset(&request, 0, sizeof(HAP_power_request_t));
- request.type = HAP_power_set_HMX;
- request.hmx.power_up = TRUE;
- FARF(ALWAYS, "Powering HMX on\n");
- err = HAP_power_set((void *) ctx, &request);
- if (err != AEE_SUCCESS) {
- FARF(ERROR, "ggml-hex: error powering on HMX.");
- return err;
- }
+AEEResult htp_iface_open(const char * uri, remote_handle64 * handle) {
+ (void) uri;
+ struct htp_handle * h = calloc(1, sizeof(*h));
+ if (h == NULL) {
+ return AEE_ENOMEMORY;
}
-#endif
+ *handle = (remote_handle64) h;
return AEE_SUCCESS;
}
AEEResult htp_iface_etm(remote_handle64 handle, uint32_t enable) {
+ struct htp_handle * h = (struct htp_handle *) handle;
+ if (!h) {
+ return AEE_EBADPARM;
+ }
+
int err = enable ? HAP_user_etm_enable() : HAP_user_etm_disable();
if (err) {
if (err == AEE_EVERSIONNOTSUPPORT) {
}
AEEResult htp_iface_profiler(remote_handle64 handle, uint32_t mode, const htp_iface_pmu_conf* pmu_conf) {
- struct htp_context * ctx = (struct htp_context *) handle;
- if (!ctx) {
+ struct htp_handle * h = (struct htp_handle *) handle;
+ if (!h || !h->ctx) {
return AEE_EBADPARM;
}
+ struct htp_context * ctx = h->ctx;
if (mode == HTP_PROF_PMU) {
const uint32_t* events = pmu_conf->events;
}
AEEResult htp_iface_close(remote_handle64 handle) {
- struct htp_context * ctx = (struct htp_context *) handle;
-
- if (!ctx) {
+ struct htp_handle * h = (struct htp_handle *) handle;
+ if (!h) {
return AEE_EBADPARM;
}
- if (ctx->queue) {
- FARF(ERROR, "Closing handle with queue still open");
- return AEE_EITEMBUSY;
- }
+ struct htp_context * ctx = h->ctx;
+ if (ctx) {
+ if (ctx->dsp_queue) {
+ FARF(ERROR, "Closing handle with queue still open");
+ return AEE_EITEMBUSY;
+ }
- // release the mmaps (if any)
- for (uint32_t i=0; i<HTP_MAX_MMAPS; i++) {
- if (ctx->mmap[i].size) {
+ // release the mmaps (if any)
+ for (uint32_t i=0; i<HTP_MAX_MMAPS; i++) {
+ if (ctx->mmap[i].size) {
#if __HVX_ARCH__ > 73
- HAP_munmap2((void *) ctx->mmap[i].base, ctx->mmap[i].size);
+ HAP_munmap2((void *) ctx->mmap[i].base, ctx->mmap[i].size);
#else
- HAP_munmap((void *) ctx->mmap[i].base, ctx->mmap[i].size);
+ HAP_munmap((void *) ctx->mmap[i].base, ctx->mmap[i].size);
#endif
- ctx->mmap[i].size = 0;
- ctx->mmap[i].base = NULL;
- ctx->mmap[i].fd = -1;
+ ctx->mmap[i].size = 0;
+ ctx->mmap[i].base = NULL;
+ ctx->mmap[i].fd = -1;
+ }
}
- }
- if (ctx->profiler) {
- qurt_pmu_enable(1);
- }
+ if (ctx->profiler) {
+ qurt_pmu_enable(1);
+ }
+
+ if (ctx->etm) {
+ HAP_user_etm_disable();
+ }
- if (ctx->etm) {
- HAP_user_etm_disable();
+ // Free the unified block (ctx is the base address of the block)
+ free(ctx);
+ h->ctx = NULL;
}
- free(ctx);
+ free(h);
return AEE_SUCCESS;
}
AEEResult htp_iface_mmap(remote_handle64 handle, uint32_t fd, uint32_t size) {
- struct htp_context * ctx = (struct htp_context *) handle;
- if (!ctx) {
+ struct htp_handle * h = (struct htp_handle *) handle;
+ if (!h || !h->ctx) {
return AEE_EBADPARM;
}
+ struct htp_context * ctx = h->ctx;
// See if we already have this mapping
for (uint32_t i=0; i<HTP_MAX_MMAPS; i++) {
}
AEEResult htp_iface_munmap(remote_handle64 handle, uint32 fd) {
- struct htp_context * ctx = (struct htp_context *) handle;
- if (!ctx) {
+ struct htp_handle * h = (struct htp_handle *) handle;
+ if (!h || !h->ctx) {
return AEE_EBADPARM;
}
+ struct htp_context * ctx = h->ctx;
for (uint32_t i=0; i<HTP_MAX_MMAPS; i++) {
struct htp_mmap *m = &ctx->mmap[i];
}
}
+static void htp_main_thread(void * context);
static void htp_packet_callback(dspqueue_t queue, int error, void * context);
static void htp_error_callback(dspqueue_t queue, int error, void * context);
AEEResult htp_iface_start(remote_handle64 handle, uint32_t sess_id, uint64_t dsp_queue_id, uint32_t n_hvx, uint32_t n_hmx, uint64_t max_vmem) {
- struct htp_context * ctx = (struct htp_context *) handle;
-
- if (!ctx) {
+ struct htp_handle * h = (struct htp_handle *) handle;
+ if (!h) {
return AEE_EBADPARM;
}
- if (ctx->queue) {
+ if (h->ctx) {
FARF(ERROR, "Queue already open");
return AEE_EITEMBUSY;
}
- // Import queue created on the CPU
- int err = dspqueue_import(dsp_queue_id, // Queue ID from dspqueue_export
- htp_packet_callback, // Packet callback
- htp_error_callback, // Error callback; no errors expected on the DSP
- (void *) ctx, // Callback context
- &ctx->queue);
+ // Cache the original FastRPC thread priority, then calculate compute priority
+ int fastrpc_tid = qurt_thread_get_id();
+ int fastrpc_prio = qurt_thread_get_priority(fastrpc_tid);
+ int main_prio = fastrpc_prio - 10;
+ if (main_prio < 1) main_prio = 1;
+
+ dspqueue_t dsp_queue = NULL;
+ bool use_callbacks = false;
+
+ // Import queue with NULL callbacks to avoid starting dspueue internal threads
+ int err = dspqueue_import(dsp_queue_id, NULL, NULL, (void *) h, &dsp_queue);
+ if (err == AEE_EBADPARM) {
+ // Fallback for devices that don't support NULL callbacks
+ FARF(HIGH, "dspqueue import with NULL callbacks failed, trying with callbacks");
+ use_callbacks = true;
+ err = dspqueue_import(dsp_queue_id, htp_packet_callback, htp_error_callback, (void *) h, &dsp_queue);
+ }
+
if (err) {
FARF(ERROR, "Queue import failed with 0x%08x", (unsigned) err);
return err;
}
+ qurt_sysenv_max_hthreads_t hw_threads;
+ qurt_sysenv_get_max_hw_threads(&hw_threads);
+ uint32_t hw_nhvx = (qurt_hvx_get_units() >> 8) & 0xFF;
+
+ if (n_hvx == 0) {
+ n_hvx = hw_nhvx;
+ }
+ if (n_hvx > hw_threads.max_hthreads) {
+ n_hvx = hw_threads.max_hthreads;
+ }
+ if (n_hvx > HTP_MAX_NTHREADS) {
+ n_hvx = HTP_MAX_NTHREADS;
+ }
+
+ // layout segments of our contiguous block
+
+ // 1. htp_context : sits at the base (block is 4K-aligned via memalign below)
+ size_t offset = sizeof(struct htp_context);
+
+ // 2. main_stack
+ size_t offset_main_stack = 0;
+ size_t size_main_stack = 0;
+ if (!use_callbacks) {
+ offset_main_stack = hex_align_up(offset, 4096);
+ size_main_stack = MAIN_THREAD_STACK_SIZE;
+ offset = offset_main_stack + size_main_stack;
+ }
+
+ // 3. work_queue
+ size_t wq_align = work_queue_alignof();
+ size_t offset_wq = hex_align_up(offset, wq_align);
+ size_t size_wq = work_queue_sizeof(n_hvx, WORK_QUEUE_CAPACITY, WORK_QUEUE_STACK_SIZE);
+ offset = offset_wq + size_wq;
+
+ // 4. dma_queue
+ size_t dma_align = dma_queue_alignof();
+ size_t offset_dma = hex_align_up(offset, dma_align);
+ size_t size_dma = 0;
+ for (uint32_t i = 0; i < n_hvx; i++) {
+ size_dma = hex_align_up(size_dma, dma_queue_alignof());
+ size_dma += dma_queue_sizeof(256);
+ size_dma = hex_align_up(size_dma, dma_queue_alignof());
+ size_dma += dma_queue_alias_sizeof();
+ }
+ offset = offset_dma + size_dma;
+
+ // 5. hmx_queue
+ size_t offset_hmx = 0;
+ size_t size_hmx = 0;
+ if (n_hmx) {
+ size_t hmx_align = hmx_queue_alignof();
+ offset_hmx = hex_align_up(offset, hmx_align);
+ size_hmx = hmx_queue_sizeof(HMX_QUEUE_CAPACITY, HMX_QUEUE_STACK_SIZE);
+ offset = offset_hmx + size_hmx;
+ }
+
+ size_t footprint = hex_align_up(offset, 128);
+
+ void * block = memalign(4096, footprint);
+ if (!block) {
+ FARF(ERROR, "Unable to allocate unified block of size %zu\n", footprint);
+ dspqueue_close(dsp_queue);
+ return AEE_ENOMEMORY;
+ }
+ memset(block, 0, footprint);
+
+ h->ctx = (struct htp_context *) block;
+ struct htp_context * ctx = h->ctx;
+ ctx->footprint = footprint;
+
+ ctx->thread_id = fastrpc_tid;
+ ctx->thread_prio = main_prio;
ctx->max_vmem = max_vmem;
- ctx->thread_id = qurt_thread_get_id();
- ctx->thread_prio = qurt_thread_get_priority(ctx->thread_id);
+ ctx->dsp_queue = dsp_queue;
- // allocate VTCM
err = vtcm_alloc(ctx);
if (err != AEE_SUCCESS) {
FARF(ERROR, "Unable to allocate VTCM");
+ htp_iface_stop(handle);
return AEE_ENOMEMORY;
}
- ctx->hmx_enabled = n_hmx;
- ctx->hmx_queue = NULL;
- if (n_hmx) {
- ctx->hmx_queue = hmx_queue_create(16, ctx->vtcm_rctx);
- if (ctx->hmx_queue) {
- ctx->hmx_queue->trace = &ctx->trace[HTP_MAX_NTHREADS];
- } else {
- FARF(ERROR, "hmx-queue-create failed");
- ctx->hmx_enabled = false;
+ HAP_setFARFRuntimeLoggingParams(0xffff, NULL, 0);
+
+ // Set client class
+ {
+ HAP_power_request_t request;
+ memset(&request, 0, sizeof(HAP_power_request_t));
+ request.type = HAP_power_set_apptype;
+ request.apptype = HAP_POWER_COMPUTE_CLIENT_CLASS;
+
+ if ((err = HAP_power_set((void *) ctx, &request)) != 0) {
+ htp_iface_stop(handle);
+ return err;
}
}
- FARF(HIGH, "HMX %s (n_hmx=%d)", ctx->hmx_enabled ? "enabled" : "disabled", n_hmx);
- qurt_sysenv_max_hthreads_t hw_threads;
- qurt_sysenv_get_max_hw_threads(&hw_threads);
- uint32_t hw_nhvx = (qurt_hvx_get_units() >> 8) & 0xFF;
+ // DCVS setup
+ {
+ HAP_power_request_t request;
+ memset(&request, 0, sizeof(request));
- if (n_hvx == 0) {
- n_hvx = hw_nhvx;
+ request.type = HAP_power_set_DCVS_v3;
+ request.dcvs_v3.set_dcvs_enable = TRUE;
+ request.dcvs_v3.dcvs_enable = FALSE;
+ request.dcvs_v3.set_bus_params = TRUE;
+ request.dcvs_v3.bus_params.min_corner = HAP_DCVS_VCORNER_MAX;
+ request.dcvs_v3.bus_params.max_corner = HAP_DCVS_VCORNER_MAX;
+ request.dcvs_v3.bus_params.target_corner = HAP_DCVS_VCORNER_MAX;
+ request.dcvs_v3.set_core_params = TRUE;
+ request.dcvs_v3.core_params.min_corner = HAP_DCVS_VCORNER_MAX;
+ request.dcvs_v3.core_params.max_corner = HAP_DCVS_VCORNER_MAX;
+ request.dcvs_v3.core_params.target_corner = HAP_DCVS_VCORNER_MAX;
+ request.dcvs_v3.set_sleep_disable = TRUE;
+ request.dcvs_v3.sleep_disable = TRUE;
+
+#if (__HEXAGON_ARCH__ >= 79)
+ HAP_set_dcvs_v3_protected_bus_corners(&request, 1);
+#endif
+ if ((err = HAP_power_set((void *) ctx, &request)) != 0) {
+ htp_iface_stop(handle);
+ return err;
+ }
+
+ memset(&request, 0, sizeof(request));
+ request.type = HAP_power_set_HVX;
+ request.hvx.power_up = TRUE;
+ if ((err = HAP_power_set((void *) ctx, &request)) != 0) {
+ htp_iface_stop(handle);
+ return err;
+ }
}
- if (n_hvx > hw_threads.max_hthreads) {
- n_hvx = hw_threads.max_hthreads;
+
+#if __HVX_ARCH__ >= 75
+ {
+ // Power on HMX and set HMX clock
+ HAP_power_request_t request;
+ memset(&request, 0, sizeof(HAP_power_request_t));
+ request.type = HAP_power_set_HMX_v2;
+ request.hmx_v2.set_power = TRUE;
+ request.hmx_v2.power_up = TRUE;
+ request.hmx_v2.set_clock = TRUE;
+ request.hmx_v2.target_corner = HAP_DCVS_EXP_VCORNER_MAX;
+ request.hmx_v2.min_corner = HAP_DCVS_EXP_VCORNER_MAX;
+ request.hmx_v2.max_corner = HAP_DCVS_EXP_VCORNER_MAX;
+ request.hmx_v2.perf_mode = HAP_CLK_PERF_HIGH;
+ FARF(ALWAYS, "Setting HMX clock\n");
+ err = HAP_power_set((void *) ctx, &request);
+ if (err != AEE_SUCCESS) {
+ FARF(ERROR, "ggml-hex: error setting HMX clock.");
+ htp_iface_stop(handle);
+ return err;
+ }
}
- if (n_hvx > HTP_MAX_NTHREADS) {
- n_hvx = HTP_MAX_NTHREADS;
+#else
+ {
+ // Power on HMX
+ HAP_power_request_t request;
+ memset(&request, 0, sizeof(HAP_power_request_t));
+ request.type = HAP_power_set_HMX;
+ request.hmx.power_up = TRUE;
+ FARF(ALWAYS, "Powering HMX on\n");
+ err = HAP_power_set((void *) ctx, &request);
+ if (err != AEE_SUCCESS) {
+ FARF(ERROR, "ggml-hex: error powering on HMX.");
+ htp_iface_stop(handle);
+ return err;
+ }
}
+#endif
+
+ ctx->hmx_enabled = n_hmx;
+ ctx->hmx_queue = NULL;
+ if (n_hmx) {
+ void * hmx_ptr = (void *) ((uintptr_t) block + offset_hmx);
+ ctx->hmx_queue = hmx_queue_init(hmx_ptr, HMX_QUEUE_CAPACITY, HMX_QUEUE_STACK_SIZE, ctx->vtcm_rctx, &ctx->trace[HTP_MAX_NTHREADS]);
+ }
+ FARF(HIGH, "HMX %s (n_hmx=%d)", ctx->hmx_enabled ? "enabled" : "disabled", n_hmx);
ctx->n_threads = n_hvx;
+ ctx->n_threads_div = init_fastdiv_values(ctx->n_threads);
+
+ // Initialize DMA queues
+ uint8_t * dma_ptr_curr = (uint8_t *) ((uintptr_t) block + offset_dma);
+ size_t size_dma_q = dma_queue_sizeof(256);
+ size_t size_dma_alias = dma_queue_alias_sizeof();
+
for (int i = 0; i < ctx->n_threads; i++) {
- ctx->dma[i] = dma_queue_create(256); // queue depth
- if (ctx->dma[i]) {
- ctx->dma[i]->trace = &ctx->trace[i];
- }
+ dma_ptr_curr = (uint8_t *) hex_align_up((uintptr_t) dma_ptr_curr, dma_queue_alignof());
+ ctx->dma_cached[i] = dma_queue_init(dma_ptr_curr, 256, (uintptr_t) ctx->vtcm_base, ctx->vtcm_size, &ctx->trace[i]);
+ dma_ptr_curr += size_dma_q;
+
+ dma_ptr_curr = (uint8_t *) hex_align_up((uintptr_t) dma_ptr_curr, dma_queue_alignof());
+ ctx->dma[i] = dma_queue_alias_init(dma_ptr_curr, ctx->dma_cached[i], 1);
+ dma_ptr_curr += size_dma_alias;
}
ctx->ddr_spad_size = 512 * 1024; // 512 KB
ctx->ddr_spad_base = memalign(128, ctx->ddr_spad_size);
- // init worker pool
- err = worker_pool_init(&ctx->worker_pool, n_hvx);
- if (err != AEE_SUCCESS) {
- FARF(ERROR, "Unable to create worker pool");
- if (ctx->ddr_spad_base) {
- free(ctx->ddr_spad_base);
- ctx->ddr_spad_base = NULL;
- ctx->ddr_spad_size = 0;
+ void * wq_ptr = (void *) ((uintptr_t) block + offset_wq);
+ ctx->work_queue = work_queue_init(wq_ptr, n_hvx, WORK_QUEUE_CAPACITY, WORK_QUEUE_STACK_SIZE);
+
+ ctx->main_stack = NULL;
+ ctx->main_thread = 0;
+ atomic_store(&ctx->killed, false);
+
+ if (!use_callbacks) {
+ // Start main compute thread
+ ctx->main_stack = (void *) ((uintptr_t) block + offset_main_stack);
+
+ qurt_thread_attr_t attr;
+ qurt_thread_attr_init(&attr);
+ qurt_thread_attr_set_stack_addr(&attr, ctx->main_stack);
+ qurt_thread_attr_set_stack_size(&attr, size_main_stack);
+ qurt_thread_attr_set_priority(&attr, main_prio);
+ qurt_thread_attr_set_name(&attr, "htp-main");
+
+ int err_thread = qurt_thread_create(&ctx->main_thread, &attr, htp_main_thread, ctx);
+ if (err_thread) {
+ FARF(ERROR, "Unable to create htp main thread: %d", err_thread);
+ htp_iface_stop(handle);
+ return AEE_ENOMEMORY;
}
- return err;
}
FARF(HIGH, "session %u started: n-hvx %u vtcm-size %zu vtcm-rctx %u n-threads %u thread-id %d thread-prio %d \n",
}
AEEResult htp_iface_stop(remote_handle64 handle) {
- struct htp_context * ctx = (struct htp_context *) handle;
- if (!ctx) {
+ struct htp_handle * h = (struct htp_handle *) handle;
+ if (!h || !h->ctx) {
return AEE_EBADPARM;
}
+ struct htp_context * ctx = h->ctx;
- if (!ctx->queue) {
- FARF(ERROR, "Queue not open");
- return AEE_EBADSTATE;
+ if (ctx->main_thread) {
+ atomic_store(&ctx->killed, true);
+ int status;
+ (void) qurt_thread_join(ctx->main_thread, &status);
+ ctx->main_thread = 0;
}
- // Close queue. dspqueue_close() will also wait for callbacks to finish.
- int err = dspqueue_close(ctx->queue);
- ctx->queue = NULL;
+ int err = dspqueue_close(ctx->dsp_queue); ctx->dsp_queue = NULL;
if (err != 0) {
FARF(ERROR, "Queue close failed with 0x%08x", (unsigned) err);
return err;
}
- if (ctx->worker_pool) {
- // Release worker pool
- worker_pool_release(&ctx->worker_pool);
- }
+ work_queue_free(ctx->work_queue);
for (int i = 0; i < ctx->n_threads; i++) {
- dma_queue_delete(ctx->dma[i]);
+ dma_queue_alias_free(ctx->dma[i]);
+ dma_queue_free(ctx->dma_cached[i]);
}
if (ctx->hmx_queue) {
- hmx_queue_delete(ctx->hmx_queue);
+ hmx_queue_free(ctx->hmx_queue);
ctx->hmx_queue = NULL;
}
ctx->hmx_enabled = false;
ctx->ddr_spad_size = 0;
}
+ free(ctx);
+ h->ctx = NULL;
+
return AEE_SUCCESS;
}
case HTP_OP_INVALID:
break;
-
- // No default to catch missing cases
}
FARF(ERROR, "Unknown Op %u", octx->op);
}
}
-static void prep_tensor(struct htp_context *ctx, struct htp_buf_desc *bufs, uint32_t idx, struct htp_tensor *t) {
+static void prep_tensor(struct htp_context *ctx, struct htp_buf_desc *bufs, struct htp_tensor *tens, uint32_t idx, struct htp_tensor *t) {
uint32_t offset = t->data;
uint32_t size = t->size;
uint32_t bi = t->bi;
+ uint32_t alias = t->alias;
- t->data = bufs[bi].base + offset; // update data to the actual pointer
+ t->data = (uint32_t) (bufs[bi].base + offset); // update data to the actual pointer
+ t->alias = (uint32_t) (tens + alias); // update alias to the actual pointer
FARF(HIGH, "prep-tensor #%u: bi %u offset %u size %u data %p : %u:%u:%u:%u", idx, t->bi, offset, t->size, (void*) t->data,
t->ne[0], t->ne[1], t->ne[3], t->ne[3]);
static void prep_tensors(struct htp_context *ctx, struct htp_buf_desc *bufs, struct htp_tensor *tens, uint32_t n_tens) {
for (uint32_t i=0; i < n_tens; i++) {
- prep_tensor(ctx, bufs, i, tens + i);
+ prep_tensor(ctx, bufs, tens, i, tens + i);
}
}
// Prep input tensors
for (uint32_t i=0; i<HTP_OP_MAX_INPUTS; i++) {
- struct htp_tensor *src = op->src[i] == 0xffff ? NULL : tens + op->src[i];
-
- octx->src[i] = src;
- if (!src) continue;
-
- if (!(src->flags & HTP_TENSOR_FLUSHED) && (src->flags & HTP_TENSOR_COMPUTE)) {
- // flush compute buffers on input
- hex_l2flush((void *) src->data, src->size);
+ uint16_t src_idx = op->src[i];
+ if (src_idx == 0xffff) {
+ octx->src[i] = NULL;
+ octx->src_dma[i] = NULL;
+ continue;
}
+ struct htp_tensor *src = tens + src_idx;
+ octx->src[i] = src;
+ octx->src_dma[i] = octx->ctx->dma; // FIXME: ? octx->ctx->dma_cached : octx->ctx->dma;
+
FARF(HIGH, "prep-src #%u: data %p size %u : %u:%u:%u:%u", op->src[i], (void*) src->data, src->size,
src->ne[0], src->ne[1], src->ne[3], src->ne[3]);
}
+ htp_tensor_flush_all(octx->ctx, octx->src, HTP_OP_MAX_INPUTS);
+
// Prep output tensors
for (uint32_t i = 0; i < HTP_OP_MAX_OUTPUTS; i++) {
uint16_t dst_idx = op->dst[i];
if (dst_idx == 0xffff) {
- octx->dsts[i] = NULL;
+ octx->dsts[i] = NULL;
+ octx->dst_dma[i] = NULL;
continue;
}
struct htp_tensor *dst = tens + dst_idx;
- octx->dsts[i] = dst;
+ octx->dsts[i] = dst;
+ octx->dst_dma[i] = octx->ctx->dma; // FIXME: ? octx->ctx->dma_cached : octx->ctx->dma;
+
+ htp_tensor_make_dirty(dst, octx->ctx->dirty_map);
FARF(HIGH, "prep-dst[%u] #%u: data %p size %u : %u:%u:%u:%u", i, dst_idx, (void*) dst->data, dst->size,
dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3]);
octx->src3_spad.src = NULL;
octx->dst_spad.src = NULL;
- // flush buffers on output
- for (uint32_t i = 0; i < HTP_OP_MAX_OUTPUTS; i++) {
- if (octx->dsts[i]) {
- struct htp_tensor *dst = (struct htp_tensor *)octx->dsts[i];
- hex_l2flush((void *) dst->data, dst->size);
- dst->flags |= HTP_TENSOR_FLUSHED;
+ return status;
+}
+
+static void process_opbatch(struct htp_context * ctx, const struct htp_opbatch_req * req, const struct dspqueue_buffer * dbuf) {
+ dspqueue_t queue = ctx->dsp_queue;
+ int err;
- FARF(HIGH, "post-dst[%u] #%u: data %p size %u : %u:%u:%u:%u", i, op->dst[i], (void*) dst->data, dst->size,
- dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3]);
+ const uint32_t n_bufs = req->n_bufs;
+ const uint32_t n_tens = req->n_tensors;
+ const uint32_t n_ops = req->n_ops;
+
+ const uint32_t b_size = sizeof(struct htp_buf_desc) * n_bufs;
+ const uint32_t t_size = sizeof(struct htp_tensor) * n_tens;
+ const uint32_t o_size = sizeof(struct htp_op_desc) * n_ops;
+ const uint32_t p_size = sizeof(struct htp_prof_desc) * n_ops;
+ const uint32_t tr_size = (HTP_MAX_NTHREADS + 1) * req->n_traces * sizeof(struct htp_trace_desc);
+
+ if (dbuf->size < b_size + t_size + o_size + p_size + tr_size) {
+ FARF(ERROR, "invalid opbatch memory block size %u (req %u)", dbuf->size, b_size + t_size + o_size + p_size + tr_size);
+ return;
+ }
+
+ FARF(HIGH, "processing opbatch #%u: n-bufs %u n-tensors %u n-ops %u n-traces %u : m-size %u b-size %u t-size %u o-size %u", req->id,
+ n_bufs, n_tens, n_ops, req->n_traces, dbuf->size, b_size, t_size, o_size);
+
+ // Clean cache at the start of the batch
+ // We cant trace this part because the trace buffer is setup later
+ qurt_mem_cache_clean((qurt_addr_t) 0, 0, QURT_MEM_CACHE_FLUSH_INVALIDATE_ALL, QURT_MEM_DCACHE);
+ hex_l2fetch_block(ctx, ctx->footprint);
+ bitmap_reset(ctx->dirty_map, HTP_OP_MAX_TENSORS);
+
+ // Setup descriptor pointers
+ uint8_t * m_ptr = dbuf->ptr;
+ struct htp_buf_desc* bufs = (struct htp_buf_desc*) m_ptr; m_ptr += b_size;
+ struct htp_tensor* tens = (struct htp_tensor*) m_ptr; m_ptr += t_size;
+ struct htp_op_desc* ops = (struct htp_op_desc*) m_ptr; m_ptr += o_size;
+ struct htp_prof_desc* pds = (struct htp_prof_desc*) m_ptr;
+
+ prep_op_bufs(ctx, bufs, n_bufs);
+ prep_tensors(ctx, bufs, tens, n_tens);
+
+ struct htp_ops_context *octx = &ctx->octx;
+ memset(octx, 0, sizeof(*octx));
+ octx->n_threads = ctx->n_threads;
+ octx->ctx = ctx;
+
+ memset(ctx->trace, 0, sizeof(ctx->trace));
+ if (ctx->profiler == HTP_PROF_TRACE) {
+ struct htp_trace_desc * trace_events = (struct htp_trace_desc *) (m_ptr + p_size);
+ for (int t = 0; t <= HTP_MAX_NTHREADS; t++) {
+ ctx->trace[t].events = &trace_events[t * req->n_traces];
+ ctx->trace[t].max_events = req->n_traces;
}
}
- return status;
+ work_queue_wakeup(ctx->work_queue);
+ if (ctx->hmx_queue) {
+ hmx_queue_wakeup(ctx->hmx_queue);
+ }
+
+ int op_status = HTP_STATUS_OK;
+ for (uint32_t i = 0; i < n_ops && op_status == HTP_STATUS_OK; i++) {
+ struct profile_data prof;
+
+ profile_start(ctx->profiler, &prof);
+
+ op_status = proc_op_req(octx, tens, i, &ops[i]);
+
+ profile_stop(ctx->profiler, &prof);
+
+ if (ctx->profiler) {
+ pds[i].opcode = ops[i].opcode;
+ pds[i].usecs = prof.usecs;
+ pds[i].cycles_start = prof.cycles_start;
+ pds[i].cycles_stop = prof.cycles_stop;
+ for (int j = 0; j < HEX_NUM_PMU_COUNTERS; j++) {
+ pds[i].pmu[j] = prof.pmu_counters[j];
+ }
+ }
+ }
+
+ if (ctx->hmx_queue) {
+ hmx_queue_suspend(ctx->hmx_queue);
+ hmx_queue_flush(ctx->hmx_queue);
+ }
+ work_queue_suspend(ctx->work_queue);
+
+ struct htp_opbatch_rsp rsp;
+ memset(&rsp, 0, sizeof(rsp));
+ rsp.id = req->id;
+ rsp.status = op_status;
+ rsp.n_bufs = n_bufs;
+ rsp.n_tensors = n_tens;
+ rsp.n_ops = n_ops;
+
+ if (ctx->profiler == HTP_PROF_TRACE) {
+ for (int t = 0; t <= HTP_MAX_NTHREADS; t++) {
+ rsp.n_traces[t] = ctx->trace[t].count;
+ }
+ }
+
+ struct dspqueue_buffer write_dbuf = *dbuf;
+ write_dbuf.flags = DSPQUEUE_BUFFER_FLAG_FLUSH_SENDER | DSPQUEUE_BUFFER_FLAG_INVALIDATE_RECIPIENT;
+
+ // Flush remaining dirty tensors at the end of the batch
+ htp_trace_event_start(&ctx->trace[0], HTP_TRACE_EVT_L2FLUSH, 0);
+ qurt_mem_cache_clean((qurt_addr_t) 0, 0, QURT_MEM_CACHE_FLUSH_INVALIDATE_ALL, QURT_MEM_DCACHE);
+ htp_trace_event_stop(&ctx->trace[0], HTP_TRACE_EVT_L2FLUSH, 0);
+
+ err = dspqueue_write(queue, 0, 1, &write_dbuf, sizeof(rsp), (const uint8_t *) &rsp, DSPQUEUE_TIMEOUT_NONE);
+ if (err != 0) {
+ FARF(ERROR, "dspqueue_write failed: 0x%08x", (unsigned) err);
+ }
}
+#define DSPQUEUE_READ_TIMEOUT_USEC 5000
#define DSPQUEUE_POLL_TIMEOUT_USEC 100
#define DSPQUEUE_POLL_COUNT 100
-static void htp_packet_callback(dspqueue_t queue, int error, void * context) {
- struct htp_context * ctx = (struct htp_context *) context;
-
+static void process_ops(struct htp_context * ctx) {
+ dspqueue_t queue = ctx->dsp_queue;
int err;
uint32_t poll_count = DSPQUEUE_POLL_COUNT;
vtcm_acquire(ctx);
- while (!ctx->vtcm_needs_release) {
+ while (!ctx->vtcm_needs_release && !atomic_load(&ctx->killed)) {
struct htp_opbatch_req req;
uint32_t r_size = sizeof(req);
// Reset poll count for valid requests
poll_count = DSPQUEUE_POLL_COUNT;
- const uint32_t n_bufs = req.n_bufs;
- const uint32_t n_tens = req.n_tensors;
- const uint32_t n_ops = req.n_ops;
-
- const uint32_t b_size = sizeof(struct htp_buf_desc) * n_bufs;
- const uint32_t t_size = sizeof(struct htp_tensor) * n_tens;
- const uint32_t o_size = sizeof(struct htp_op_desc) * n_ops;
- const uint32_t p_size = sizeof(struct htp_prof_desc) * n_ops;
- const uint32_t tr_size = (HTP_MAX_NTHREADS + 1) * req.n_traces * sizeof(struct htp_trace_desc);
-
- if (dbuf.size < b_size + t_size + o_size + p_size + tr_size) {
- FARF(ERROR, "invalid opbatch memory block size %u (req %u)", dbuf.size, b_size + t_size + o_size + p_size + tr_size);
- break;
- }
-
- FARF(HIGH, "processing opbatch #%u: n-bufs %u n-tensors %u n-ops %u n-traces %u : m-size %u b-size %u t-size %u o-size %u", req.id,
- n_bufs, n_tens, n_ops, req.n_traces, dbuf.size, b_size, t_size, o_size);
-
- // Setup descriptor pointers
- uint8_t * m_ptr = dbuf.ptr;
- struct htp_buf_desc* bufs = (struct htp_buf_desc*) m_ptr; m_ptr += b_size;
- struct htp_tensor* tens = (struct htp_tensor*) m_ptr; m_ptr += t_size;
- struct htp_op_desc* ops = (struct htp_op_desc*) m_ptr; m_ptr += o_size;
- struct htp_prof_desc* pds = (struct htp_prof_desc*) m_ptr;
-
- prep_op_bufs(ctx, bufs, n_bufs);
- prep_tensors(ctx, bufs, tens, n_tens);
-
- struct htp_ops_context *octx = &ctx->octx;
- memset(octx, 0, sizeof(*octx));
- octx->n_threads = ctx->n_threads;
- octx->ctx = ctx;
-
- if (ctx->profiler == HTP_PROF_TRACE) {
- memset(ctx->trace, 0, sizeof(ctx->trace));
- struct htp_trace_desc * trace_events = (struct htp_trace_desc *) (m_ptr + p_size);
- for (int t = 0; t <= HTP_MAX_NTHREADS; t++) {
- ctx->trace[t].events = &trace_events[t * req.n_traces];
- ctx->trace[t].max_events = req.n_traces;
- }
- } else {
- for (int t = 0; t <= HTP_MAX_NTHREADS; t++) {
- ctx->trace[t].events = NULL;
- ctx->trace[t].max_events = 0;
- }
- }
-
- int op_status = HTP_STATUS_OK;
- uint32_t op_wakeup = n_ops / 2; // half-way throgh the batch
-
- hmx_queue_wakeup(ctx->hmx_queue);
-
- for (uint32_t i=0; i < n_ops; i++) {
- struct profile_data prof;
-
- if (i == op_wakeup) {
- dspqueue_write_early_wakeup_noblock(queue, 0, 0);
- }
+ process_opbatch(ctx, &req, &dbuf);
+ }
- profile_start(ctx->profiler, &prof);
+ vtcm_release(ctx);
+}
- op_status = proc_op_req(octx, tens, i, &ops[i]);
+static void htp_packet_callback(dspqueue_t queue, int error, void * context) {
+ (void) queue;
+ (void) error;
+ struct htp_handle * h = (struct htp_handle *) context;
+ if (h && h->ctx) {
+ process_ops(h->ctx);
+ }
+}
- profile_stop(ctx->profiler, &prof);
+static void htp_main_thread(void * context) {
+ struct htp_context * ctx = (struct htp_context *) context;
- if (op_status != HTP_STATUS_OK) {
- break;
- }
+ FARF(HIGH, "htp-main-thread: started");
- if (ctx->profiler) {
- pds[i].opcode = ops[i].opcode;
- pds[i].usecs = prof.usecs;
- pds[i].cycles_start = prof.cycles_start;
- pds[i].cycles_stop = prof.cycles_stop;
- for (int j = 0; j < HEX_NUM_PMU_COUNTERS; j++) {
- pds[i].pmu[j] = prof.pmu_counters[j];
- }
- }
- }
+ while (!atomic_load(&ctx->killed)) {
+ uint32_t flags = 0;
+ uint32_t num_buffers = 0;
+ uint32_t message_length = 0;
- hmx_queue_suspend(ctx->hmx_queue);
-
- struct htp_opbatch_rsp rsp;
- rsp.id = req.id;
- rsp.status = op_status;
- rsp.n_bufs = n_bufs;
- rsp.n_tensors = n_tens;
- rsp.n_ops = n_ops;
- memset(rsp.pad, 0, sizeof(rsp.pad));
- if (ctx->profiler == HTP_PROF_TRACE) {
- for (int t = 0; t <= HTP_MAX_NTHREADS; t++) {
- rsp.n_traces[t] = ctx->trace[t].count;
- }
+ int err = dspqueue_peek(ctx->dsp_queue, &flags, &num_buffers, &message_length, 50000);
+ if (err == 0) {
+ process_ops(ctx);
+ } else if (err == AEE_EWOULDBLOCK || err == AEE_EEXPIRED) {
+ continue;
} else {
- memset(rsp.n_traces, 0, sizeof(rsp.n_traces));
- }
-
- dbuf.flags = DSPQUEUE_BUFFER_FLAG_FLUSH_SENDER | DSPQUEUE_BUFFER_FLAG_INVALIDATE_RECIPIENT;
-
- err = dspqueue_write(queue, 0, 1, &dbuf, sizeof(rsp), (const uint8_t *) &rsp, DSPQUEUE_TIMEOUT_NONE);
- if (err != 0) {
- FARF(ERROR, "dspqueue_write failed: 0x%08x", (unsigned) err);
+ FARF(ERROR, "dspqueue_peek failed: 0x%08x", (unsigned) err);
break;
}
}
- vtcm_release(ctx);
+ FARF(HIGH, "htp-main-thread: stopped");
}
// Per thread quant tasks
// Precomputed block-parallel quantization values
worker_callback_t quant_task_func;
- uint32_t quant_ib_first[MAX_NUM_WORKERS];
- uint32_t quant_ib_last[MAX_NUM_WORKERS];
- uint32_t quant_r[MAX_NUM_WORKERS];
- uint32_t quant_c[MAX_NUM_WORKERS];
+ uint32_t quant_ib_first[WORK_QUEUE_MAX_N_THREADS];
+ uint32_t quant_ib_last[WORK_QUEUE_MAX_N_THREADS];
+ uint32_t quant_r[WORK_QUEUE_MAX_N_THREADS];
+ uint32_t quant_c[WORK_QUEUE_MAX_N_THREADS];
uint32_t n_quant_tasks;
uint32_t n_quant_rows_per_thread;
atomic_uint quant_barrier;
return;
}
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL;
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0_start);
const uint32_t blck_0 = 64;
const uint32_t src0_start_row = src0_nrows_per_thread * ith; \
const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); \
\
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL; \
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \
\
const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \
const uint32_t n_prefetch = kparams->n_prefetch; \
const uint32_t src0_start_row = src0_nrows_per_thread * ith; \
const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); \
\
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL; \
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \
\
const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \
const uint32_t n_prefetch = kparams->n_prefetch; \
uint8_t * restrict vtcm_src3_ptr = mmctx->vtcm_src3 + mmctx->vtcm_src3_size_per_thread * ith; \
uint8_t * restrict src1_data = mmctx->vtcm_src1; \
\
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL; \
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \
\
const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \
const uint32_t n_prefetch = kparams->n_prefetch; \
uint8_t * restrict vtcm_src2_ptr = mmctx->vtcm_src2 + mmctx->vtcm_src2_size_per_thread * ith; \
uint8_t * restrict src1_data = mmctx->vtcm_src1; \
\
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL; \
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \
\
const uint8_t * restrict src0_row = (const uint8_t *) src0->data; \
const uint8_t * restrict src2_row = (const uint8_t *) src2->data; \
return; \
} \
\
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL; \
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_QUANT, ir_first); \
\
uint8_t * restrict dst = mmctx->vtcm_src1; \
static void quantize_f32_q8_0_tiled_block(unsigned int nth, unsigned int ith, void * data) {
struct htp_mm_context * mmctx = data;
struct htp_ops_context * octx = mmctx->octx;
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL;
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_QUANT, mmctx->quant_ib_first[ith]);
const struct htp_tensor * src = octx->src[1];
static void quantize_f32_q8_1_tiled_block(unsigned int nth, unsigned int ith, void * data) {
struct htp_mm_context * mmctx = data;
struct htp_ops_context * octx = mmctx->octx;
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL;
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_QUANT, mmctx->quant_ib_first[ith]);
const struct htp_tensor * src = octx->src[1];
const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows);
const uint32_t src0_end_row_x2 = src0_start_row + ((src0_end_row - src0_start_row) & ~1U);
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL;
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith];
const size_t dst_row_size = nb1;
const size_t src0_row_size = nb01;
const uint32_t src0_start_row = src0_nrows_per_thread * ith;
const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows);
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL;
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith];
const size_t dst_row_size = nb1;
const size_t src0_row_size = nb01;
return;
}
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL;
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith];
const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params;
const uint32_t n_prefetch = kparams->n_prefetch;
return;
}
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL;
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith];
const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params;
const uint32_t n_prefetch = kparams->n_prefetch;
static int hvx_mm_matmul(struct htp_ops_context * octx) {
htp_matmul_tensors_preamble;
+ struct htp_thread_trace * tr = &octx->ctx->trace[0];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0);
+
struct htp_mm_context mmctx_struct = {0};
struct htp_mm_context * mmctx = &mmctx_struct;
mmctx->octx = octx;
mmctx->vtcm_src0_stride = src0_row_size_padded;
mmctx->vtcm_src1_stride = src1_row_size;
- if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)
- return HTP_STATUS_OK;
-
if (need_quant) {
mmctx->n_quant_rows_per_thread = (src1_nrows + n_quant_tasks - 1) / n_quant_tasks;
mmctx->quant_task_func = quant_task_func;
mmctx->n_quant_tasks = 0;
}
- const uint32_t n_matmul_jobs = octx->n_threads;
- worker_pool_run_func(octx->ctx->worker_pool, matmul_job_func, mmctx, n_matmul_jobs);
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0);
+
+ worker_pool_run_func(octx->ctx->worker_pool, matmul_job_func, mmctx, octx->n_threads);
return HTP_STATUS_OK;
}
#define DEQUANTIZE_WORKER_LOOP_IMPL(SUFFIX) \
static void dequantize_tiled_worker_loop_##SUFFIX(unsigned int n, unsigned int i, void *data) { \
tiled_dequantize_state_t *state = (tiled_dequantize_state_t *)data; \
- struct htp_thread_trace * tr = state->traces ? &state->traces[i] : NULL; \
+ struct htp_thread_trace * tr = &state->traces[i]; \
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_W_DEQUANT, i); \
for (unsigned int task_id = i; task_id < (unsigned int)state->n_tasks; task_id += n) { \
int start = task_id * state->n_tiles_per_task; \
static void convert_f16_worker_loop(unsigned int n, unsigned int i, void *data) {
tiled_dequantize_state_t *state = (tiled_dequantize_state_t *)data;
- struct htp_thread_trace * tr = state->traces ? &state->traces[i] : NULL;
+ struct htp_thread_trace * tr = &state->traces[i];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_W_DEQUANT, i);
for (unsigned int task_id = i; task_id < (unsigned int)state->n_tasks; task_id += n) {
int start = task_id * state->n_tiles_per_task;
static void quantize_f32_worker_loop(unsigned int n, unsigned int i, void *data) {
tiled_dequantize_state_t *state = (tiled_dequantize_state_t *)data;
- struct htp_thread_trace * tr = state->traces ? &state->traces[i] : NULL;
+ struct htp_thread_trace * tr = &state->traces[i];
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_QUANT, i);
for (unsigned int task_id = i; task_id < (unsigned int)state->n_tasks; task_id += n) {
static void transfer_output_chunk_worker_fn(unsigned int n, unsigned int i, void *data) {
output_transfer_task_state_t *st = (output_transfer_task_state_t *) data;
- struct htp_thread_trace * tr = st->traces ? &st->traces[i] : NULL;
+ struct htp_thread_trace * tr = &st->traces[i];
int start_chunk_idx = i * st->n_chunks_per_task;
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_O_PROC, start_chunk_idx);
uint32_t dma_step_rows_shift;
} activation_transfer_task_state_t;
+typedef struct {
+ __fp16 *dst;
+ const float *src;
+ uint32_t n_rows;
+ uint32_t k_block;
+ uint32_t k_stride;
+ uint32_t k_valid;
+ uint32_t n_col_chunks;
+ struct fastdiv_values n_threads_div;
+ float *vtcm_f32_act;
+ size_t vtcm_f32_act_bytes;
+ struct htp_thread_trace *traces;
+ struct htp_context *ctx;
+ uint32_t dma_step_rows;
+ uint32_t dma_step_rows_shift;
+} activation_transfer_col_chunk_state_t;
+
+static void transfer_activation_chunk_fp32_to_fp16_dma_pipelined_col_chunk(
+ dma_queue *dma_q,
+ __fp16 *restrict vtcm_dst,
+ const float *restrict src,
+ uint32_t n_rows,
+ uint32_t k_block,
+ uint32_t k_stride,
+ uint32_t k_chunk_valid,
+ uint32_t c_first,
+ uint32_t c_len,
+ float *thread_f32_act,
+ struct htp_thread_trace *tr,
+ uint32_t dma_step_rows,
+ uint32_t dma_step_rows_shift) {
+
+ const uint32_t R = dma_step_rows;
+ const uint32_t n_rows_padded = hex_align_up(n_rows, HTP_MM_HMX_TILE_N_ROWS);
+
+ const uint32_t n_steps = n_rows_padded >> dma_step_rows_shift;
+
+ // Push step 0
+ if (n_steps > 0 && n_rows > 0) {
+ uint32_t nrows_to_fetch = hex_smin(n_rows, R);
+ dma_queue_push(dma_q, dma_make_ptr(thread_f32_act, src + c_first),
+ c_len * sizeof(float), k_stride * sizeof(float), k_chunk_valid * sizeof(float), nrows_to_fetch);
+ }
+ // Push step 1
+ if (n_steps > 1) {
+ uint32_t next_r = R * 1;
+ if (next_r < n_rows) {
+ uint32_t nrows_to_fetch = hex_smin(n_rows - next_r, R);
+ const float *next_src = src + next_r * k_stride + c_first;
+ float *next_buf = thread_f32_act + 1 * R * c_len;
+ dma_queue_push(dma_q, dma_make_ptr(next_buf, next_src),
+ c_len * sizeof(float), k_stride * sizeof(float), k_chunk_valid * sizeof(float), nrows_to_fetch);
+ }
+ }
+ for (uint32_t s = 0; s < n_steps; ++s) {
+ uint32_t r = s << dma_step_rows_shift;
+ float *curr_buf = thread_f32_act;
+
+ if (r < n_rows) {
+ curr_buf = (float *) dma_queue_pop(dma_q).dst;
+ }
+
+ htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_PREP, r);
+ for (uint32_t p = 0; p < (R >> 1); ++p) {
+ uint32_t row_idx = r + (p << 1);
+ float *pair_buf = curr_buf + (p << 1) * c_len;
+ bool r0_valid = ((row_idx + 0) < n_rows);
+ bool r1_valid = ((row_idx + 1) < n_rows);
+
+ transfer_activation_row_pair_fp32_to_fp16_col_chunk(
+ vtcm_dst, pair_buf, pair_buf + c_len, row_idx, k_block, c_first, c_len, k_chunk_valid, r0_valid, r1_valid
+ );
+ }
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_A_PREP, r);
+
+ // Push step s + 2
+ uint32_t next_s = s + 2;
+ uint32_t next_r = next_s << dma_step_rows_shift;
+ if (next_r < n_rows) {
+ uint32_t nrows_to_fetch = hex_smin(n_rows - next_r, R);
+ const float *next_src = src + next_r * k_stride + c_first;
+ dma_queue_push(dma_q, dma_make_ptr(curr_buf, next_src),
+ c_len * sizeof(float), k_stride * sizeof(float), k_chunk_valid * sizeof(float), nrows_to_fetch);
+ }
+ }
+}
+
+static void transfer_activation_chunk_fp32_to_fp16_col_chunk(
+ __fp16 *restrict vtcm_dst,
+ const float *restrict src,
+ uint32_t n_rows,
+ uint32_t k_block,
+ uint32_t k_stride,
+ uint32_t c_first,
+ uint32_t c_len,
+ uint32_t k_chunk_valid) {
+ const uint32_t n_rows_padded = hex_align_up(n_rows, HTP_MM_HMX_TILE_N_ROWS);
+ const uint32_t n_rows_tiled = (n_rows / HTP_MM_HMX_TILE_N_ROWS) * HTP_MM_HMX_TILE_N_ROWS;
+
+ uint32_t r = 0;
+
+ #pragma unroll(2)
+ for (r = 0; r < n_rows_tiled; r += 2) {
+ const float *ptr_in0 = src + (r + 0) * k_stride + c_first;
+ const float *ptr_in1 = src + (r + 1) * k_stride + c_first;
+
+ transfer_activation_row_pair_fp32_to_fp16_col_chunk(
+ vtcm_dst, ptr_in0, ptr_in1, r, k_block, c_first, c_len, k_chunk_valid, true, true
+ );
+ }
+
+ for (; r < n_rows_padded; r += 2) {
+ const bool row0_valid = r < n_rows;
+ const bool row1_valid = (r + 1) < n_rows;
+
+ const float *ptr_in0 = row0_valid ? (src + (r + 0) * k_stride + c_first) : NULL;
+ const float *ptr_in1 = row1_valid ? (src + (r + 1) * k_stride + c_first) : NULL;
+
+ transfer_activation_row_pair_fp32_to_fp16_col_chunk(
+ vtcm_dst, ptr_in0, ptr_in1, r, k_block, c_first, c_len, k_chunk_valid, row0_valid, row1_valid
+ );
+ }
+}
+
+static void transfer_activation_chunk_col_chunk_worker_fn(unsigned int n, unsigned int i, void *data) {
+ activation_transfer_col_chunk_state_t *st = (activation_transfer_col_chunk_state_t *) data;
+ struct htp_thread_trace * tr = &st->traces[i];
+
+ uint32_t n_blocks = st->k_block / 32;
+ uint32_t b_first = fastdiv(n_blocks * i, &st->n_threads_div);
+ uint32_t b_last = fastdiv(n_blocks * (i + 1), &st->n_threads_div);
+ uint32_t c_first = b_first * 32;
+ uint32_t c_last = b_last * 32;
+ uint32_t c_len = c_last - c_first;
+
+ if (c_len == 0) {
+ return;
+ }
+
+ uint32_t k_chunk_valid = 0;
+ if (st->k_valid > c_first) {
+ k_chunk_valid = hex_smin(st->k_valid, c_last) - c_first;
+ }
+
+ __fp16 *dst = st->dst;
+ const float *src = st->src;
+
+ if (st->vtcm_f32_act) {
+ size_t thread_scratch_bytes = hex_align_down(fastdiv(st->vtcm_f32_act_bytes, &st->n_threads_div), 128);
+ float *thread_f32_act = (float *)((char *)st->vtcm_f32_act + i * thread_scratch_bytes);
+
+ transfer_activation_chunk_fp32_to_fp16_dma_pipelined_col_chunk(
+ st->ctx->dma[i], dst, src, st->n_rows, st->k_block, st->k_stride, k_chunk_valid,
+ c_first, c_len, thread_f32_act, tr, st->dma_step_rows, st->dma_step_rows_shift
+ );
+ } else {
+ htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_PREP, c_first);
+ transfer_activation_chunk_fp32_to_fp16_col_chunk(
+ dst, src, st->n_rows, st->k_block, st->k_stride, c_first, c_len, k_chunk_valid
+ );
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_A_PREP, c_first);
+ }
+}
+
static void transfer_activation_chunk_fp32_to_fp16_dma_pipelined(
dma_queue *dma_q,
__fp16 *restrict vtcm_dst,
static void transfer_activation_chunk_worker_fn(unsigned int n, unsigned int i, void *data) {
activation_transfer_task_state_t *st = (activation_transfer_task_state_t *) data;
- struct htp_thread_trace * tr = st->traces ? &st->traces[i] : NULL;
+ struct htp_thread_trace * tr = &st->traces[i];
for (unsigned int task_id = i; task_id < (unsigned int)st->n_tasks; task_id += n) {
int chunk_idx = task_id * st->n_chunks_per_task;
static void transfer_activation_chunk_gathered_worker_fn(unsigned int n, unsigned int i, void *data) {
activation_transfer_gathered_task_state_t *st = data;
- struct htp_thread_trace * tr = st->traces ? &st->traces[i] : NULL;
+ struct htp_thread_trace * tr = &st->traces[i];
int chunk_idx = i;
int chunk_size = st->n_chunks_per_task;
- int start_row = st->start_row + chunk_idx * chunk_size;
+ int vtcm_start_row = chunk_idx * chunk_size;
+ int start_row = st->start_row + vtcm_start_row;
int n_rows = hex_smin(st->cne1 - start_row, chunk_size);
if (n_rows > 0) {
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_PREP, chunk_idx);
transfer_activation_chunk_fp32_to_fp16_gathered(
- st->dst, st->src, start_row, n_rows, st->k_block,
+ st->dst, st->src, start_row, vtcm_start_row, n_rows, st->k_block,
st->matrix_rows, st->cur_a, st->mapping_stride,
st->ne11, &st->ne11_div, st->nb11, st->nb12, st->cne1, st->k_valid);
htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_A_PREP, chunk_idx);
static void transfer_activation_chunk_gathered_worker_flat_fn(unsigned int n, unsigned int i, void *data) {
activation_transfer_gathered_task_state_t *st = data;
- struct htp_thread_trace * tr = st->traces ? &st->traces[i] : NULL;
+ struct htp_thread_trace * tr = &st->traces[i];
int chunk_idx = i;
int chunk_size = st->n_chunks_per_task;
- int start_row = st->start_row + chunk_idx * chunk_size;
+ int vtcm_start_row = chunk_idx * chunk_size;
+ int start_row = st->start_row + vtcm_start_row;
int n_rows = hex_smin(st->cne1 - start_row, chunk_size);
if (n_rows > 0) {
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_PREP, chunk_idx);
transfer_activation_chunk_fp32_to_fp16_gathered_flat(
- st->dst, st->src, start_row, n_rows, st->k_block,
+ st->dst, st->src, start_row, vtcm_start_row, n_rows, st->k_block,
st->matrix_rows, st->cur_a, st->mapping_stride,
st->nb12, st->cne1, st->k_valid);
htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_A_PREP, chunk_idx);
static void transfer_output_chunk_scattered_worker_fn(unsigned int n, unsigned int i, void *data) {
output_transfer_scattered_task_state_t *st = data;
- struct htp_thread_trace * tr = st->traces ? &st->traces[i] : NULL;
+ struct htp_thread_trace * tr = &st->traces[i];
int chunk_idx = i;
int chunk_size = st->n_chunks_per_task;
- int start_row = st->start_row + chunk_idx * chunk_size;
+ int vtcm_start_row = chunk_idx * chunk_size;
+ int start_row = st->start_row + vtcm_start_row;
int n_rows = hex_smin(st->cne1 - start_row, chunk_size);
if (n_rows > 0) {
htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_O_PROC, chunk_idx);
transfer_output_chunk_fp16_to_fp32_scattered(
- st->dst, st->vtcm_src, start_row, n_rows, st->n_cols,
+ st->dst, st->vtcm_src, start_row, vtcm_start_row, n_rows, st->n_cols,
st->matrix_rows, st->cur_a, st->mapping_stride,
st->dst_nb1, st->dst_nb2, st->cne1);
htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_O_PROC, chunk_idx);
}
}
-static void transfer_activation_chunk_threaded(
- struct htp_context *ctx,
- __fp16 *dst,
- const float *src,
- int n_rows,
- int k_block,
- int k_stride,
- int n_threads,
- int k_valid,
- float *vtcm_f32_act,
- size_t vtcm_f32_act_bytes) {
+struct activation_transfer_params {
+ struct htp_context * ctx;
+ __fp16 * dst;
+ const float * src;
+ int n_rows;
+ int k_block;
+ int k_stride;
+ int n_threads;
+ const struct fastdiv_values * act_threads_div;
+ const struct fastdiv_values * k_div;
+ int k_valid;
+ float * vtcm_f32_act;
+ size_t vtcm_f32_act_bytes;
+};
+
+static void transfer_activation_chunk_threaded(const struct activation_transfer_params * params) {
+ struct htp_context * ctx = params->ctx;
+ __fp16 * dst = params->dst;
+ const float * src = params->src;
+ int n_rows = params->n_rows;
+ int k_block = params->k_block;
+ int k_stride = params->k_stride;
+ int n_threads = params->n_threads;
+ const struct fastdiv_values * act_threads_div = params->act_threads_div;
+ const struct fastdiv_values * k_div = params->k_div;
+ int k_valid = params->k_valid;
+ float * vtcm_f32_act = params->vtcm_f32_act;
+ size_t vtcm_f32_act_bytes = params->vtcm_f32_act_bytes;
+
if (n_rows <= 0) {
return;
}
+ const size_t n_tasks = (n_rows + 31) >> 5;
+ if (n_threads > 1 && k_block > 32 && n_tasks < (size_t) n_threads) {
+ // Calculate step rows parameters for column-chunked dma pipelining
+ uint32_t dma_step_rows = 2;
+ uint32_t dma_step_rows_shift = 1;
+ if (vtcm_f32_act && vtcm_f32_act_bytes > 0 && k_block > 0) {
+ size_t thread_scratch_bytes = hex_align_down(fastdiv(vtcm_f32_act_bytes, act_threads_div), 128);
+ size_t thread_scratch_elements = thread_scratch_bytes / sizeof(float);
+ size_t dma_step_rows_max = fastdiv(thread_scratch_elements / 2, k_div);
+ if (dma_step_rows_max >= 4) {
+ dma_step_rows = 4;
+ dma_step_rows_shift = 2;
+ }
+ }
+
+ activation_transfer_col_chunk_state_t col_state;
+ col_state.dst = dst;
+ col_state.src = src;
+ col_state.n_rows = n_rows;
+ col_state.k_block = k_block;
+ col_state.k_stride = k_stride;
+ col_state.k_valid = k_valid;
+ col_state.n_col_chunks = n_threads;
+ col_state.n_threads_div = *act_threads_div;
+ col_state.vtcm_f32_act = vtcm_f32_act;
+ col_state.vtcm_f32_act_bytes = vtcm_f32_act_bytes;
+ col_state.traces = ctx->trace;
+ col_state.ctx = ctx;
+ col_state.dma_step_rows = dma_step_rows;
+ col_state.dma_step_rows_shift = dma_step_rows_shift;
+
+ worker_pool_run_func(ctx->worker_pool, transfer_activation_chunk_col_chunk_worker_fn, &col_state, n_threads);
+ return;
+ }
+
assert(k_block % HTP_MM_HMX_TILE_N_COLS == 0 && k_stride % HTP_MM_HMX_TILE_N_COLS == 0);
size_t n_tot_chunks = n_rows;
size_t n_chunks_per_task = (n_threads == 1) ? n_tot_chunks : 32; // must be multiple of 32 to ensure correct destination address
- uint32_t dma_step_rows = 2;
- uint32_t dma_step_rows_shift = 1;
- if (vtcm_f32_act && vtcm_f32_act_bytes > 0 && k_block > 0) {
- size_t thread_scratch_elements = vtcm_f32_act_bytes / (n_threads * sizeof(float));
- size_t dma_step_rows_max = (thread_scratch_elements / 2) / k_block;
- if (dma_step_rows_max >= 4) {
- dma_step_rows = 4;
- dma_step_rows_shift = 2;
- } else {
- dma_step_rows = 2;
- dma_step_rows_shift = 1;
- }
- }
-
activation_transfer_task_state_t state;
- state.n_tasks = (n_tot_chunks + n_chunks_per_task - 1) / n_chunks_per_task;
+ state.n_tasks = (n_threads == 1) ? 1 : hmx_ceil_div(n_tot_chunks, 32);
state.n_tot_chunks = n_tot_chunks;
state.n_chunks_per_task = n_chunks_per_task;
state.dst = dst;
state.vtcm_f32_act = vtcm_f32_act;
int active_threads = hex_smin(n_threads, (int)state.n_tasks);
- state.vtcm_f32_act_bytes_per_thread = (vtcm_f32_act_bytes / active_threads) & ~127u;
+ state.vtcm_f32_act_bytes_per_thread = hex_align_down(vtcm_f32_act_bytes / active_threads, 128);
+
+ uint32_t dma_step_rows = 2;
+ uint32_t dma_step_rows_shift = 1;
+ if (vtcm_f32_act && state.vtcm_f32_act_bytes_per_thread > 0 && k_block > 0) {
+ size_t thread_scratch_elements = state.vtcm_f32_act_bytes_per_thread / sizeof(float);
+ size_t dma_step_rows_max = fastdiv(thread_scratch_elements / 2, k_div);
+ if (dma_step_rows_max >= 4) {
+ dma_step_rows = 4;
+ dma_step_rows_shift = 2;
+ }
+ }
state.dma_step_rows = dma_step_rows;
state.dma_step_rows_shift = dma_step_rows_shift;
int pipeline,
int n_threads,
int act_threads,
+ const struct fastdiv_values * act_threads_div,
+ const struct fastdiv_values * k_div,
int tile_size,
int aligned_tile_size,
int vtcm_size) {
+ struct htp_thread_trace * tr = &ctx->trace[0];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0);
+
if (k % 32 != 0 || n % 32 != 0) { return -1; }
if (!hex_is_aligned(dst, VLEN) || !hex_is_aligned(activation, VLEN)) { return -1; }
int n_chunk_cnt = hmx_ceil_div(n, n_chunk_n_cols);
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0);
+
if (pipeline) {
// --- Asynchronous Pipelined Loop ---
hmx_matmul_job_t job_slots[2]; // persistent double-buffered job descriptors
void *vtcm_weight_bufs[2] = { vtcm_scratch0, vtcm_scratch1 };
void *vtcm_output_bufs[2] = { vtcm_output, vtcm_scratch2 };
- transfer_activation_chunk_threaded(ctx, vtcm_f16_act, activation + mr * act_stride, n_rows, k, act_stride, act_threads, k_valid, vtcm_f32_act, L.act_f32_bytes);
+ struct activation_transfer_params act_params = {
+ .ctx = ctx,
+ .dst = vtcm_f16_act,
+ .src = activation + mr * act_stride,
+ .n_rows = (int) n_rows,
+ .k_block = k,
+ .k_stride = act_stride,
+ .n_threads = act_threads,
+ .act_threads_div = act_threads_div,
+ .k_div = k_div,
+ .k_valid = k_valid,
+ .vtcm_f32_act = vtcm_f32_act,
+ .vtcm_f32_act_bytes = L.act_f32_bytes,
+ };
+ transfer_activation_chunk_threaded(&act_params);
// Prologue: push A0 and optionally A1 (if n_chunk_cnt > 1)
const size_t n_cols_A0 = hex_smin(n - 0 * n_chunk_n_cols, n_chunk_n_cols);
for (size_t mr = 0; mr < m; mr += m_chunk_n_rows) {
const size_t n_rows = hex_smin(m - mr, m_chunk_n_rows);
- transfer_activation_chunk_threaded(ctx, vtcm_f16_act, activation + mr * act_stride, n_rows, k, act_stride, act_threads, k_valid, vtcm_f32_act, L.act_f32_bytes);
+ struct activation_transfer_params act_params = {
+ .ctx = ctx,
+ .dst = vtcm_f16_act,
+ .src = activation + mr * act_stride,
+ .n_rows = (int) n_rows,
+ .k_block = k,
+ .k_stride = act_stride,
+ .n_threads = act_threads,
+ .act_threads_div = act_threads_div,
+ .k_div = k_div,
+ .k_valid = k_valid,
+ .vtcm_f32_act = vtcm_f32_act,
+ .vtcm_f32_act_bytes = L.act_f32_bytes,
+ };
+ transfer_activation_chunk_threaded(&act_params);
// A0: Pre-fetch the first weight chunk (nc = 0)
if (n > 0) {
static int hmx_mm_f16_f32_batched_simple(struct htp_context *ctx,
const hmx_mm_f16_f32_batched_params_t *params,
- int m_chunk, int n_chunk, int pipeline, int n_threads, int act_threads, int vtcm_size) {
+ int m_chunk, int n_chunk, int pipeline, int n_threads, int act_threads, int vtcm_size,
+ const struct fastdiv_values * act_threads_div, const struct fastdiv_values * k_div) {
int ret = 0;
for (int b3 = 0; b3 < params->ne13 && ret == 0; ++b3) {
for (int b2 = 0; b2 < params->ne12 && ret == 0; ++b2) {
params->act_stride, params->weight_stride * (int)sizeof(__fp16),
HTP_TYPE_F16, params->k, params->dst_stride, params->src2_stride, params->n,
m_chunk, n_chunk, pipeline, n_threads, act_threads,
- 0, 0, vtcm_size);
+ act_threads_div, k_div, 0, 0, vtcm_size);
}
}
return ret;
}
static int hmx_mm_f16_f32_batched(struct htp_context *ctx, const hmx_mm_f16_f32_batched_params_t *params,
- int m_chunk, int n_chunk, int pipeline, int n_threads, int act_threads, int vtcm_size) {
+ int m_chunk, int n_chunk, int pipeline, int n_threads, int act_threads,
+ const struct fastdiv_values * act_threads_div,
+ const struct fastdiv_values * k_div,
+ int vtcm_size) {
if (params->act_stride < params->k || params->weight_stride < params->k || params->dst_stride < params->n) { return -1; }
if (params->ne02 <= 0 || params->ne03 <= 0 || params->ne12 <= 0 || params->ne13 <= 0) { return -1; }
if (params->ne12 % params->ne02 != 0 || params->ne13 % params->ne03 != 0) { return -1; }
// Grouped path is only valid if group_size > 1 and it fits within VTCM budget.
bool run_grouped = (group_size > 1 && (size_t)vtcm_size <= vtcm_budget);
if (!run_grouped) {
- return hmx_mm_f16_f32_batched_simple(ctx, params, m_chunk, n_chunk, pipeline, n_threads, act_threads, vtcm_size);
+ return hmx_mm_f16_f32_batched_simple(ctx, params, m_chunk, n_chunk, pipeline, n_threads, act_threads, vtcm_size, act_threads_div, k_div);
}
+ struct htp_thread_trace * tr = &ctx->trace[0];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0);
+
const size_t vec_dot_size = params->k * sizeof(__fp16);
const bool use_dma_activation = (params->act_stride > params->k);
if (L.total_bytes > vtcm_budget) {
FARF(HIGH, "%s: grouped layout overflowed VTCM, falling back to simple batched loop", __func__);
- return hmx_mm_f16_f32_batched_simple(ctx, params, m_chunk, n_chunk, pipeline, n_threads, act_threads, vtcm_size);
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0);
+ return hmx_mm_f16_f32_batched_simple(ctx, params, m_chunk, n_chunk, pipeline, n_threads, act_threads, vtcm_size, act_threads_div, k_div);
}
uint8_t * const base = (uint8_t *) ctx->vtcm_base;
const size_t fp16_row_bytes = (size_t) params->k * sizeof(__fp16);
const size_t weight_row_bytes = (size_t) params->weight_stride * sizeof(__fp16);
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0);
+
hmx_matmul_job_t job;
for (int b3 = 0; b3 < params->ne13; ++b3) {
for (int g = 0; g < group_size; ++g) {
const float *activation_chunk = hmx_mm_activation_batch_ptr(params, b2_base + g, b3) + mr * params->act_stride;
__fp16 *vtcm_act_g = vtcm_f16_act + (size_t) g * L.act_head_stride;
- transfer_activation_chunk_threaded(ctx, vtcm_act_g,
- activation_chunk, (int) n_rows,
- params->k, params->act_stride, act_threads, params->k, vtcm_f32_act, L.act_f32_bytes);
+ struct activation_transfer_params act_params = {
+ .ctx = ctx,
+ .dst = vtcm_act_g,
+ .src = activation_chunk,
+ .n_rows = (int) n_rows,
+ .k_block = params->k,
+ .k_stride = params->act_stride,
+ .n_threads = act_threads,
+ .act_threads_div = act_threads_div,
+ .k_div = k_div,
+ .k_valid = params->k,
+ .vtcm_f32_act = vtcm_f32_act,
+ .vtcm_f32_act_bytes = L.act_f32_bytes,
+ };
+ transfer_activation_chunk_threaded(&act_params);
}
// Prologue: Push A0 and A1 (if exists)
const struct mmid_row_mapping *matrix_rows,
int cur_a,
int mapping_stride) {
+ struct htp_thread_trace * tr = &ctx->trace[0];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0);
+
const int cne1 = m;
const int m_padded = hex_align_up(m, 32);
hmx_init_column_scales(vtcm_scales, Q6_V_vsplat_R(0x3c00));
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0);
+
hmx_matmul_job_t job;
for (size_t mr = 0; mr < (size_t) m_padded; mr += m_chunk_n_rows) {
const int act_stride = (int)(src1->nb[1] / sizeof(float));
const int wgt_stride = (int)(src0->nb[1] / sizeof(__fp16));
- if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) {
- return HTP_STATUS_OK;
- }
-
const float * src2_ptr = NULL;
uint32_t src2_stride = 0;
size_t src2_nb2 = 0;
kparams->m_chunk, kparams->n_chunk,
kparams->pipeline, n_threads,
kparams->n_act_threads,
+ &kparams->div_n_act_threads,
+ &kparams->div_ne00_padded,
kparams->vtcm_size);
} else {
ret = hmx_mm_2d_f32(
(int)(dst->nb[1] / sizeof(float)), src2_stride, (int)dst->ne[0],
kparams->m_chunk, kparams->n_chunk, kparams->pipeline, n_threads,
kparams->n_act_threads,
+ &kparams->div_n_act_threads,
+ &kparams->div_ne00_padded,
kparams->tile_size, kparams->aligned_tile_size, kparams->vtcm_size
);
}
bool must_free_mapping
) {
htp_matmul_tensors_preamble;
+
+ struct htp_thread_trace * tr = &octx->ctx->trace[0];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0);
+
const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params;
const struct htp_tensor * restrict ids = octx->src[2];
const size_t src0_row_size = nb01;
mmctx->n_quant_tasks = n_quant_tasks;
atomic_init(&mmctx->quant_barrier, n_quant_tasks);
- const uint32_t n_matmul_jobs = octx->n_threads;
- worker_pool_run_func(octx->ctx->worker_pool, matmul_id_job_func, mmctx, n_matmul_jobs);
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0);
+
+ worker_pool_run_func(octx->ctx->worker_pool, matmul_id_job_func, mmctx, octx->n_threads);
if (must_free_mapping) free(mapping_buf);
return HTP_STATUS_OK;
int op_matmul_id(struct htp_ops_context * octx) {
htp_matmul_tensors_preamble;
+ struct htp_thread_trace * tr = &octx->ctx->trace[0];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0);
+
struct htp_mm_context mmctx_struct = {0};
struct htp_mm_context * mmctx = &mmctx_struct;
mmctx->octx = octx;
}
}
- if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) {
- if (must_free_mapping) free(mapping_buf);
- return HTP_STATUS_OK;
- }
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0);
if (kparams->n_hmx) {
return hmx_mm_op_matmul_id(octx, mmctx, matrix_row_counts, matrix_rows, mapping_buf, must_free_mapping);
}
int op_matmul_qkv(struct htp_ops_context * octx) {
+ struct htp_thread_trace * tr = &octx->ctx->trace[0];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0);
+
const struct htp_tensor * restrict src0 = octx->src[0]; // Wk
const struct htp_tensor * restrict src1 = octx->src[1]; // x
const struct htp_tensor * restrict src2 = octx->src[2]; // Wv
mmctx->vtcm_src3_size_per_thread = L.src3_bytes / octx->n_threads;
mmctx->vtcm_dst_size_per_thread = L.dst_bytes / octx->n_threads;
- if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)
- return HTP_STATUS_OK;
-
mmctx->n_quant_rows_per_thread = (src1_nrows + n_quant_tasks - 1) / n_quant_tasks;
mmctx->quant_task_func = quant_task_func;
mmctx->n_quant_tasks = n_quant_tasks;
} else {
matmul_job_func = hvx_mm_qkv_2d;
}
+
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0);
+
worker_pool_run_func(octx->ctx->worker_pool, matmul_job_func, mmctx, n_matmul_jobs);
return HTP_STATUS_OK;
}
int op_matmul_ffn(struct htp_ops_context * octx) {
+ struct htp_thread_trace * tr = &octx->ctx->trace[0];
+ htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0);
+
const struct htp_tensor * restrict src0 = octx->src[0]; // Wgate
const struct htp_tensor * restrict src1 = octx->src[1]; // y
const struct htp_tensor * restrict src2 = octx->src[2]; // Wup
mmctx->vtcm_src2_size_per_thread = L.src2_bytes / octx->n_threads;
mmctx->vtcm_dst_size_per_thread = L.dst_bytes / octx->n_threads;
- if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)
- return HTP_STATUS_OK;
-
mmctx->n_quant_rows_per_thread = (src1_nrows + n_quant_tasks - 1) / n_quant_tasks;
mmctx->quant_task_func = quant_task_func;
mmctx->n_quant_tasks = n_quant_tasks;
} else {
matmul_job_func = hvx_mm_ffn_2d;
}
+
+ htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0);
+
worker_pool_run_func(octx->ctx->worker_pool, matmul_job_func, mmctx, n_matmul_jobs);
return HTP_STATUS_OK;
struct fastdiv_values div_r2;
struct fastdiv_values div_r3;
struct fastdiv_values div_ne11;
+ struct fastdiv_values div_n_act_threads;
+ struct fastdiv_values div_ne00_padded;
};
#if defined(__cplusplus)
return L.total_bytes;
}
+static inline bool htp_mm_hmx_solve_batched_params(
+ int wtype,
+ uint32_t k,
+ uint32_t ne01_padded,
+ uint32_t ne11,
+ uint32_t group_size,
+ bool use_dma_activation,
+ int n_threads,
+ bool pipeline,
+ size_t vtcm_budget,
+ size_t * m_chunk_out,
+ size_t * n_chunk_out,
+ int * act_threads_out,
+ size_t * vtcm_size_out
+) {
+ size_t best_mblocks = SIZE_MAX;
+ int best_act_threads = 0;
+ size_t best_m_chunk = 0;
+ size_t best_n_chunk = 0;
+ size_t best_vtcm_size = 0;
+
+ int act_threads = n_threads;
+ while (act_threads >= 1) {
+ size_t group_overhead = 256;
+ size_t group_size_per_n, group_size_per_m, group_size_per_mn;
+ htp_mm_hmx_get_batched_chunk_costs(k, group_size, &group_size_per_n, &group_size_per_m, &group_size_per_mn);
+
+ size_t m_chunk_candidate = 0;
+ size_t n_chunk_candidate = 0;
+ size_t vtcm_size_candidate = 0;
+
+ if (htp_mm_hmx_compute_chunks(vtcm_budget, group_overhead, group_size_per_n, group_size_per_m, group_size_per_mn, hex_align_up(ne11, 32), ne01_padded,
+ (size_t) ne01_padded * HTP_MM_HMX_COST_W_DEQUANT, (size_t) ne11 * HTP_MM_HMX_COST_A_CONVERT,
+ &m_chunk_candidate, &n_chunk_candidate, &vtcm_size_candidate) == 0) {
+ size_t exact_size = htp_mm_hmx_get_batched_vtcm_size(wtype, k, m_chunk_candidate, n_chunk_candidate, group_size, use_dma_activation, pipeline, act_threads);
+ if (exact_size <= vtcm_budget) {
+ size_t mblocks = ((size_t) ne11 + m_chunk_candidate - 1) / m_chunk_candidate;
+ if (mblocks < best_mblocks || (mblocks == best_mblocks && act_threads > best_act_threads)) {
+ best_mblocks = mblocks;
+ best_act_threads = act_threads;
+ best_m_chunk = m_chunk_candidate;
+ best_n_chunk = n_chunk_candidate;
+ best_vtcm_size = exact_size;
+ }
+ }
+ }
+ if (act_threads == 1) {
+ act_threads = 0;
+ } else {
+ act_threads /= 2;
+ }
+ }
+
+ if (best_act_threads > 0) {
+ *m_chunk_out = best_m_chunk;
+ *n_chunk_out = best_n_chunk;
+ *vtcm_size_out = best_vtcm_size;
+ *act_threads_out = best_act_threads;
+ return true;
+ }
+ return false;
+}
+
+static inline bool htp_mm_hmx_solve_2d_params(
+ int wtype,
+ uint32_t k,
+ uint32_t m_id_rows,
+ uint32_t ne01_padded,
+ uint32_t ne11_padded,
+ uint32_t m_for_cost,
+ int n_threads,
+ bool pipeline,
+ bool is_matmul_id,
+ uint32_t aligned_tile_size,
+ size_t vtcm_budget,
+ size_t * m_chunk_out,
+ size_t * n_chunk_out,
+ int * act_threads_out,
+ size_t * vtcm_size_out
+) {
+ size_t best_mblocks = SIZE_MAX;
+ int best_act_threads = 0;
+ size_t best_m_chunk = 0;
+ size_t best_n_chunk = 0;
+ size_t best_vtcm_size = 0;
+
+ const int m_for_chunks = is_matmul_id ? hex_align_up(m_id_rows, 32) : ne11_padded;
+
+ int act_threads = n_threads;
+ while (act_threads >= 1) {
+ size_t simple_2d_overhead = 256;
+ size_t simple_2d_size_per_n, simple_2d_size_per_m, simple_2d_size_per_mn;
+ htp_mm_hmx_get_2d_chunk_costs(wtype, k, pipeline, aligned_tile_size, &simple_2d_size_per_n, &simple_2d_size_per_m, &simple_2d_size_per_mn);
+
+ size_t m_chunk_candidate = 0;
+ size_t n_chunk_candidate = 0;
+ size_t vtcm_size_candidate = 0;
+
+ if (htp_mm_hmx_compute_chunks(vtcm_budget, simple_2d_overhead, simple_2d_size_per_n, simple_2d_size_per_m, simple_2d_size_per_mn, m_for_chunks, ne01_padded,
+ (size_t) ne01_padded * HTP_MM_HMX_COST_W_DEQUANT, (size_t) m_for_cost * HTP_MM_HMX_COST_A_CONVERT,
+ &m_chunk_candidate, &n_chunk_candidate, &vtcm_size_candidate) == 0) {
+ size_t exact_size = htp_mm_hmx_get_2d_vtcm_size(wtype, k, m_chunk_candidate, n_chunk_candidate, pipeline, is_matmul_id ? 0 : act_threads, aligned_tile_size);
+ if (exact_size <= vtcm_budget) {
+ size_t mblocks = ((size_t) m_for_cost + m_chunk_candidate - 1) / m_chunk_candidate;
+ if (mblocks < best_mblocks || (mblocks == best_mblocks && act_threads > best_act_threads)) {
+ best_mblocks = mblocks;
+ best_act_threads = act_threads;
+ best_m_chunk = m_chunk_candidate;
+ best_n_chunk = n_chunk_candidate;
+ best_vtcm_size = exact_size;
+ }
+ }
+ }
+ if (act_threads == 1) {
+ act_threads = 0;
+ } else {
+ act_threads /= 2;
+ }
+ }
+
+ if (best_act_threads > 0) {
+ *m_chunk_out = best_m_chunk;
+ *n_chunk_out = best_n_chunk;
+ *vtcm_size_out = best_vtcm_size;
+ *act_threads_out = best_act_threads;
+ return true;
+ }
+ return false;
+}
+
#ifdef __cplusplus
}
#endif
#include "htp-ctx.h"
#include "htp-ops.h"
#include "htp-ops.h"
+#include "htp-tensor.h"
// Redefined the rope type constants as we can't include ggml.h
#define HTP_ROPE_TYPE_NORMAL 0
}
int op_rope(struct htp_ops_context * octx) {
- int err = HTP_STATUS_OK;
-
switch (octx->src[0]->type) {
case HTP_TYPE_F32:
- err = execute_op_rope_f32(octx);
- break;
+ return execute_op_rope_f32(octx);
default:
- err = HTP_STATUS_NO_SUPPORT;
- break;
+ return HTP_STATUS_NO_SUPPORT;
}
-
- return err;
}
#include "ggml-common.h"
#include "htp-ctx.h"
#include "htp-ops.h"
+#include "htp-tensor.h"
#include "htp-vtcm.h"
#include "hex-profile.h"
struct htp_ops_context * octx = uctx->octx; \
const struct htp_tensor * src = octx->src[0]; \
const struct htp_tensor * dst = octx->dst; \
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL; \
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \
\
htp_unary_preamble; \
\
struct htp_ops_context * octx = uctx->octx; \
const struct htp_tensor * src = octx->src[0]; \
const struct htp_tensor * dst = octx->dst; \
- struct htp_thread_trace * tr = octx->ctx ? &octx->ctx->trace[ith] : NULL; \
+ struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \
\
htp_unary_preamble; \
\
}
int op_unary(struct htp_ops_context * octx) {
- int err = HTP_STATUS_OK;
-
switch (octx->src[0]->type) {
case HTP_TYPE_F32:
- err = execute_op_unary_f32(octx);
- break;
+ return execute_op_unary_f32(octx);
default:
- err = HTP_STATUS_NO_SUPPORT;
- break;
+ return HTP_STATUS_NO_SUPPORT;
}
-
- return err;
}
--- /dev/null
+#include "work-queue.h"
+#include "hex-utils.h"
+
+#include <qurt.h>
+#include <qurt_hvx.h>
+
+#include <stdatomic.h>
+#include <stdint.h>
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+
+#include "HAP_farf.h"
+
+#define LOWEST_USABLE_QURT_PRIO (254)
+
+// internal structure kept in thread-local storage per instance of work queue
+typedef struct {
+ work_queue_t queue;
+ unsigned int id;
+} worker_context_t;
+
+struct work_queue_task_s {
+ work_queue_func_t func;
+ void * data;
+ unsigned int n_threads;
+ atomic_uint barrier;
+};
+
+// internal structure kept in thread-local storage per instance of work queue
+struct work_queue_s {
+ atomic_uint seqn; // seqno used to detect new jobs
+ atomic_uint idx_read; // Updated by producer (pop/reclaim)
+ unsigned int idx_write; // Updated by producer (push)
+ uint32_t idx_mask;
+ uint32_t capacity;
+
+ qurt_thread_t thread[WORK_QUEUE_MAX_N_THREADS]; // thread ID's of the workers
+ worker_context_t context[WORK_QUEUE_MAX_N_THREADS]; // worker contexts
+ void * stack[WORK_QUEUE_MAX_N_THREADS]; // thread stack pointers
+ unsigned int n_threads; // total threads (workers + main)
+ unsigned int n_workers; // number of active threads (just workers)
+
+ atomic_bool active; // workers are polling/active
+ atomic_bool killed; // threads need to exit
+ bool external_mem; // memory owned externally
+
+ struct work_queue_task_s queue[] __attribute__((aligned(HEX_L2_LINE_SIZE)));
+};
+
+static void work_queue_thread(void * context) {
+ worker_context_t * me = (worker_context_t *) context;
+ work_queue_t q = me->queue;
+
+ FARF(HIGH, "work-queue: thread %u started", me->id);
+
+ unsigned int prev_seqn = 0;
+
+ while (!atomic_load_explicit(&q->killed, memory_order_relaxed)) {
+ unsigned int seqn = atomic_load_explicit(&q->seqn, memory_order_acquire);
+ if (seqn == prev_seqn) {
+ if (atomic_load_explicit(&q->active, memory_order_relaxed)) {
+ hex_pause();
+ } else {
+ qurt_futex_wait(&q->seqn, prev_seqn);
+ }
+ continue;
+ }
+
+ prev_seqn = seqn;
+
+ // Process all active tasks in the queue
+ unsigned int ir = atomic_load_explicit(&q->idx_read, memory_order_relaxed);
+ unsigned int iw = q->idx_write;
+
+ while (ir != iw) {
+ struct work_queue_task_s * task = &q->queue[ir];
+
+ unsigned int n = task->n_threads;
+ unsigned int i = me->id;
+ if (i < n) {
+ task->func(n, i, task->data);
+
+ atomic_fetch_sub_explicit(&task->barrier, 1, memory_order_release);
+ } else {
+ while (atomic_load_explicit(&task->barrier, memory_order_relaxed) > 0) {
+ hex_pause();
+ }
+ }
+
+ ir = (ir + 1) & q->idx_mask;
+ }
+ }
+
+ FARF(HIGH, "work-queue: thread %u stopped", me->id);
+}
+
+bool work_queue_run_async(work_queue_t q, work_queue_func_t func, void * data, unsigned int n) {
+ if (n > q->n_threads) {
+ FARF(ERROR, "work-queue: invalid number of jobs %u for n-threads %u", n, q->n_threads);
+ return false;
+ }
+
+ unsigned int ir = atomic_load_explicit(&q->idx_read, memory_order_relaxed);
+ unsigned int iw = q->idx_write;
+
+ if (((iw + 1) & q->idx_mask) == ir) {
+ FARF(ERROR, "work-queue-push: queue is full\n");
+ return false;
+ }
+
+ struct work_queue_task_s * task = &q->queue[iw];
+ task->func = func;
+ task->data = data;
+ task->n_threads = n;
+ atomic_store_explicit(&task->barrier, n, memory_order_relaxed);
+
+ q->idx_write = (iw + 1) & q->idx_mask;
+
+ // publish job to workers (already awake and polling)
+ atomic_fetch_add_explicit(&q->seqn, 1, memory_order_release);
+
+ // main thread runs job #0
+ func(n, 0, data);
+
+ atomic_fetch_sub_explicit(&task->barrier, 1, memory_order_release);
+
+ while (atomic_load_explicit(&task->barrier, memory_order_relaxed) > 0) {
+ hex_pause();
+ }
+
+ atomic_thread_fence(memory_order_acquire);
+
+ atomic_store_explicit(&q->idx_read, (ir + 1) & q->idx_mask, memory_order_relaxed);
+
+ return true;
+}
+
+size_t work_queue_sizeof(uint32_t n_threads, uint32_t capacity, uint32_t stack_size) {
+ capacity = hex_ceil_pow2(capacity);
+ uint32_t n_workers = n_threads > 1 ? n_threads - 1 : 0;
+ size_t size_stacks = stack_size * n_workers;
+ size_t size_q = hex_align_up(sizeof(struct work_queue_s) + capacity * sizeof(struct work_queue_task_s), HEX_L2_LINE_SIZE);
+ return size_stacks + size_q;
+}
+
+size_t work_queue_alignof(void) {
+ return 4096;
+}
+
+work_queue_t work_queue_init(void * ptr, uint32_t n_threads, uint32_t capacity, uint32_t stack_size) {
+ capacity = hex_ceil_pow2(capacity);
+ uint32_t n_workers = n_threads > 1 ? n_threads - 1 : 0;
+ unsigned char * mem_blob = (unsigned char *) ptr;
+
+ work_queue_t q = (work_queue_t) (mem_blob + stack_size * n_workers);
+ memset(q, 0, sizeof(struct work_queue_s) + capacity * sizeof(struct work_queue_task_s));
+
+ q->n_threads = n_threads;
+ q->n_workers = n_workers;
+ q->external_mem = true;
+ q->capacity = capacity;
+
+ for (unsigned int i = 0; i < n_workers; i++) {
+ q->stack[i] = mem_blob; mem_blob += stack_size;
+ q->thread[i] = 0;
+ q->context[i].id = i + 1;
+ q->context[i].queue = q;
+ }
+
+ atomic_init(&q->idx_read, 0);
+ atomic_init(&q->seqn, 0);
+ atomic_init(&q->active, false);
+ q->idx_write = 0;
+ q->idx_mask = capacity - 1;
+ q->killed = 0;
+ for (int i = 0; i < (int) capacity; i++) {
+ atomic_init(&q->queue[i].barrier, 0);
+ q->queue[i].func = NULL;
+ q->queue[i].data = NULL;
+ q->queue[i].n_threads = 0;
+ }
+
+ // launch the workers
+ qurt_thread_attr_t attr;
+ qurt_thread_attr_init(&attr);
+
+ for (unsigned int i = 0; i < n_workers; i++) {
+ qurt_thread_attr_set_stack_addr(&attr, q->stack[i]);
+ qurt_thread_attr_set_stack_size(&attr, stack_size);
+
+ char thread_name[32];
+ snprintf(thread_name, sizeof(thread_name), "work-queue:%u", i);
+ qurt_thread_attr_set_name(&attr, thread_name);
+
+ // set up priority - by default, match the creating thread's prio
+ int prio = qurt_thread_get_priority(qurt_thread_get_id());
+ if (prio < 1) {
+ prio = 1;
+ }
+ if (prio > LOWEST_USABLE_QURT_PRIO) {
+ prio = LOWEST_USABLE_QURT_PRIO;
+ }
+
+ qurt_thread_attr_set_priority(&attr, prio);
+
+ int err = qurt_thread_create(&q->thread[i], &attr, work_queue_thread, (void *) &q->context[i]);
+ if (err) {
+ FARF(ERROR, "Could not launch worker threads!");
+ work_queue_free(q);
+ return NULL;
+ }
+ }
+
+ return q;
+}
+
+void work_queue_free(work_queue_t q) {
+ if (!q) { return; }
+
+ atomic_store_explicit(&q->killed, 1, memory_order_relaxed);
+ atomic_fetch_add_explicit(&q->seqn, 1, memory_order_release);
+ qurt_futex_wake(&q->seqn, q->n_workers);
+
+ for (unsigned int i = 0; i < q->n_workers; i++) {
+ if (q->thread[i]) {
+ int status;
+ (void) qurt_thread_join(q->thread[i], &status);
+ }
+ }
+}
+
+void work_queue_wakeup(work_queue_t q) {
+ if (!atomic_load_explicit(&q->active, memory_order_relaxed)) {
+ atomic_store_explicit(&q->active, true, memory_order_release);
+ // Increment seqn and wake workers to transition them out of sleep
+ atomic_fetch_add_explicit(&q->seqn, 1, memory_order_release);
+ qurt_futex_wake(&q->seqn, q->n_workers);
+ }
+}
+
+void work_queue_suspend(work_queue_t q) {
+ atomic_store_explicit(&q->active, false, memory_order_release);
+}
--- /dev/null
+#ifndef HTP_WORK_QUEUE_H
+#define HTP_WORK_QUEUE_H
+
+#include <stdbool.h>
+#include <stdint.h>
+#include <stddef.h>
+
+typedef void (*work_queue_func_t)(unsigned int n, unsigned int i, void *);
+
+struct work_queue_s;
+typedef struct work_queue_s * work_queue_t;
+
+#define WORK_QUEUE_MAX_N_THREADS 10
+
+size_t work_queue_sizeof(uint32_t n_threads, uint32_t capacity, uint32_t stack_size);
+size_t work_queue_alignof(void);
+work_queue_t work_queue_init(void * ptr, uint32_t n_threads, uint32_t capacity, uint32_t stack_size);
+void work_queue_free(work_queue_t q);
+
+void work_queue_wakeup(work_queue_t q);
+void work_queue_suspend(work_queue_t q);
+
+bool work_queue_run_async(work_queue_t q, work_queue_func_t func, void * data, unsigned int n);
+
+static inline bool work_queue_run(work_queue_t q, work_queue_func_t func, void * data, unsigned int n) {
+ if (n <= 1) {
+ func(n, 0, data);
+ return true;
+ }
+ return work_queue_run_async(q, func, data, n);
+}
+
+// Legacy compatibility
+typedef work_queue_func_t worker_callback_t;
+#define worker_pool_run_func work_queue_run
+#define worker_pool work_queue
+
+#endif // #ifndef HTP_WORK_QUEUE_H
+++ /dev/null
-#include "worker-pool.h"
-#include "hex-utils.h"
-
-#include <qurt.h>
-#include <qurt_hvx.h>
-
-#include <stdatomic.h>
-#include <stdint.h>
-#include <stdio.h>
-#include <stdlib.h>
-#include <string.h>
-
-#include "HAP_farf.h"
-
-#define LOWEST_USABLE_QURT_PRIO (254)
-
-struct worker_pool_s;
-
-// internal structure kept in thread-local storage per instance of worker pool
-typedef struct {
- struct worker_pool_s * pool;
- unsigned int id;
-} worker_context_t;
-
-// internal structure kept in thread-local storage per instance of worker pool
-typedef struct worker_pool_s {
- worker_pool_job_t job[MAX_NUM_WORKERS]; // list of job descriptors
- qurt_thread_t thread[MAX_NUM_WORKERS]; // thread ID's of the workers
- worker_context_t context[MAX_NUM_WORKERS]; // worker contexts
- void * stack[MAX_NUM_WORKERS]; // thread stack pointers
- unsigned int n_threads; // number of workers in this pool
-
- atomic_uint seqn; // seqno used to detect new jobs
- atomic_uint next_job; // next job index
- atomic_uint n_pending; // number of pending jobs
- atomic_uint n_jobs; // number of current jobs
- atomic_bool killed; // threads need to exit
-} worker_pool_t;
-
-static void worker_pool_main(void * context) {
- worker_context_t * me = (worker_context_t *) context;
- worker_pool_t * pool = me->pool;
-
- FARF(HIGH, "worker-pool: thread %u started", me->id);
-
- unsigned int prev_seqn = 0;
- unsigned int poll_cnt = WORKER_POOL_POLL_COUNT;
- while (!atomic_load(&pool->killed)) {
- unsigned int seqn = atomic_load(&pool->seqn);
- if (seqn == prev_seqn) {
- // drop HVX context while spinning
- if (poll_cnt > 1 && poll_cnt == WORKER_POOL_POLL_COUNT) {
- qurt_hvx_unlock();
- }
- if (--poll_cnt) {
- hex_pause();
- continue;
- }
- qurt_futex_wait(&pool->seqn, prev_seqn);
- poll_cnt = WORKER_POOL_POLL_COUNT;
- continue;
- }
-
- prev_seqn = seqn;
- poll_cnt = WORKER_POOL_POLL_COUNT;
-
- // New job
- unsigned int n = atomic_load(&pool->n_jobs);
- unsigned int i = atomic_fetch_add(&pool->next_job, 1);
- if (i >= n) {
- // Spurious wakeup
- continue;
- }
-
- pool->job[i].func(n, i, pool->job[i].data);
-
- atomic_fetch_sub(&pool->n_pending, 1);
- }
-
- FARF(HIGH, "worker-pool: thread %u stopped", me->id);
-}
-
-AEEResult worker_pool_init_with_stack_size(worker_pool_context_t * context, uint32_t n_threads, uint32_t stack_size) {
- int err = 0;
-
- if (NULL == context) {
- FARF(ERROR, "NULL context passed to worker_pool_init().");
- return AEE_EBADPARM;
- }
-
- // Allocations
- int size = (stack_size * n_threads) + (sizeof(worker_pool_t));
-
- unsigned char * mem_blob = (unsigned char *) malloc(size);
- if (!mem_blob) {
- FARF(ERROR, "Could not allocate memory for worker pool!!");
- return AEE_ENOMEMORY;
- }
-
- worker_pool_t * me = (worker_pool_t *) (mem_blob + stack_size * n_threads);
-
- // name for the first worker, useful in debugging threads
- char name[19];
- snprintf(name, 12, "0x%8x:", (int) me);
- strcat(name, "worker0");
- me->n_threads = n_threads;
-
- // initializations
- for (unsigned int i = 0; i < me->n_threads; i++) {
- me->stack[i] = NULL;
- me->thread[i] = 0;
-
- me->context[i].id = i;
- me->context[i].pool = me;
- }
-
- // initialize job queue
- me->n_pending = 0;
- me->n_jobs = 0;
- me->next_job = 0;
- me->seqn = 0;
- me->killed = 0;
-
- // launch the workers
- qurt_thread_attr_t attr;
- qurt_thread_attr_init(&attr);
-
- for (unsigned int i = 0; i < me->n_threads; i++) {
- // set up stack
- me->stack[i] = mem_blob;
- mem_blob += stack_size;
- qurt_thread_attr_set_stack_addr(&attr, me->stack[i]);
- qurt_thread_attr_set_stack_size(&attr, stack_size);
-
- // set up name
- qurt_thread_attr_set_name(&attr, name);
- name[17] = (name[17] + 1);
- // name threads context:worker0, context:worker1, .. (recycle at 9, but num threads should be less than that anyway)
- if (name[17] > '9') {
- name[17] = '0';
- }
-
- // set up priority - by default, match the creating thread's prio
- int prio = qurt_thread_get_priority(qurt_thread_get_id());
-
- if (prio < 1) {
- prio = 1;
- }
- if (prio > LOWEST_USABLE_QURT_PRIO) {
- prio = LOWEST_USABLE_QURT_PRIO;
- }
-
- qurt_thread_attr_set_priority(&attr, prio);
-
- // launch
- err = qurt_thread_create(&me->thread[i], &attr, worker_pool_main, (void *) &me->context[i]);
- if (err) {
- FARF(ERROR, "Could not launch worker threads!");
- worker_pool_release((worker_pool_context_t *) &me);
- return AEE_EQURTTHREADCREATE;
- }
- }
- *context = (worker_pool_context_t *) me;
- return AEE_SUCCESS;
-}
-
-AEEResult worker_pool_init(worker_pool_context_t * context, uint32_t n_threads) {
- return worker_pool_init_with_stack_size(context, n_threads, WORKER_THREAD_STACK_SZ);
-}
-
-// clean up worker pool
-void worker_pool_release(worker_pool_context_t * context) {
- worker_pool_t * me = (worker_pool_t *) *context;
-
- // if no worker pool exists, return error.
- if (NULL == me) {
- return;
- }
-
- atomic_store(&me->killed, 1);
- atomic_fetch_add(&me->seqn, 1);
- qurt_futex_wake(&me->seqn, me->n_threads);
-
- // de-initializations
- for (unsigned int i = 0; i < me->n_threads; i++) {
- if (me->thread[i]) {
- int status;
- (void) qurt_thread_join(me->thread[i], &status);
- }
- }
-
- // free allocated memory (were allocated as a single buffer starting at stack[0])
- if (me->stack[0]) {
- free(me->stack[0]);
- }
-
- *context = NULL;
-}
-
-// run jobs
-AEEResult worker_pool_run_jobs(worker_pool_context_t context, worker_pool_job_t * job, unsigned int n) {
- worker_pool_t * me = (worker_pool_t *) context;
- if (NULL == me) {
- FARF(ERROR, "worker-pool: invalid context");
- return AEE_EBADPARM;
- }
-
- if (n > me->n_threads) {
- FARF(ERROR, "worker-pool: invalid number of jobs %u for n-threads %u", n, me->n_threads);
- return AEE_EBADPARM;
- }
-
- memcpy(me->job, job, sizeof(worker_pool_job_t) * n);
-
- if (n > 1) {
- atomic_store(&me->next_job, 1);
- atomic_store(&me->n_jobs, n);
- atomic_store(&me->n_pending, n - 1);
-
- // wake up workers
- atomic_fetch_add(&me->seqn, 1);
- qurt_futex_wake(&me->seqn, n - 1);
- }
-
- // main thread runs job #0
- me->job[0].func(n, 0, me->job[0].data);
-
- if (n > 1) {
- while (atomic_load(&me->n_pending))
- ;
- }
-
- return 0;
-}
-
-// run func
-AEEResult worker_pool_run_func(worker_pool_context_t context, worker_callback_t func, void * data, unsigned int n) {
- worker_pool_job_t job[n];
-
- for (unsigned int i = 0; i < n; i++) {
- job[i].func = func;
- job[i].data = data;
- }
-
- return worker_pool_run_jobs(context, job, n);
-}
-
-AEEResult worker_pool_set_thread_priority(worker_pool_context_t context, unsigned int prio) {
- worker_pool_t * me = (worker_pool_t *) context;
-
- // if no worker pool exists, return error.
- if (!me) {
- return AEE_ENOMORE;
- }
-
- int result = AEE_SUCCESS;
- if (prio < 1) {
- prio = 1;
- }
- if (prio > LOWEST_USABLE_QURT_PRIO) {
- prio = LOWEST_USABLE_QURT_PRIO;
- }
-
- for (unsigned int i = 0; i < me->n_threads; i++) {
- int res = qurt_thread_set_priority(me->thread[i], (unsigned short) prio);
- if (0 != res) {
- result = AEE_EBADPARM;
- FARF(ERROR, "QURT failed to set priority of thread %d, ERROR = %d", me->thread[i], res);
- }
- }
-
- return result;
-}
-
-AEEResult worker_pool_retrieve_thread_id(worker_pool_context_t context, unsigned int * tids) {
- worker_pool_t * me = (worker_pool_t *) context;
- if (!me) {
- FARF(ERROR, "worker-pool: invalid context");
- return AEE_EBADPARM;
- ;
- }
-
- for (int i = 0; i < me->n_threads; i++) {
- tids[i] = me->thread[i];
- }
-
- return AEE_SUCCESS;
-}
-
-AEEResult worker_pool_get_thread_priority(worker_pool_context_t context, unsigned int * prio) {
- worker_pool_t * me = (worker_pool_t *) context;
- if (!me) {
- FARF(ERROR, "worker-pool: invalid context");
- return AEE_EBADPARM;
- }
-
- int priority = qurt_thread_get_priority(me->thread[0]);
- if (priority > 0) {
- *prio = priority;
- return 0;
- } else {
- *prio = 0;
- return AEE_EBADSTATE;
- }
-}
+++ /dev/null
-#ifndef HTP_WORKER_POOL_H
-#define HTP_WORKER_POOL_H
-
-// MACRO enables function to be visible in shared-library case.
-#define WORKERPOOL_API __attribute__((visibility("default")))
-
-#include <AEEStdDef.h>
-#include <AEEStdErr.h>
-#include <stdint.h>
-
-#ifdef __cplusplus
-extern "C" {
-#endif
-
-/// signature of callbacks to be invoked by worker threads
-typedef void (*worker_callback_t)(unsigned int n, unsigned int i, void *);
-
-/// Typedef of worker_pool context
-typedef void * worker_pool_context_t;
-
-/// descriptor for requested callback
-typedef struct {
- worker_callback_t func;
- void * data;
-} worker_pool_job_t;
-
-#define WORKER_THREAD_STACK_SZ (2 * 16384)
-
-/// Maximum supported number of worker threads.
-#define MAX_NUM_WORKERS 10
-
-#if __HVX_ARCH__ > 79
-#define WORKER_POOL_POLL_COUNT 2000
-#else
-#define WORKER_POOL_POLL_COUNT 1
-#endif
-
-// Initialize worker pool.
-WORKERPOOL_API AEEResult worker_pool_init(worker_pool_context_t * context, uint32_t n_threads);
-
-// Initialize worker pool with custom stack size
-WORKERPOOL_API AEEResult worker_pool_init_with_stack_size(worker_pool_context_t * context,
- uint32_t n_threads,
- uint32_t stack_size);
-
-// Kill worker threads and release worker pool resources
-WORKERPOOL_API void worker_pool_release(worker_pool_context_t * context);
-
-// Run jobs with the worker pool.
-WORKERPOOL_API AEEResult worker_pool_run_jobs(worker_pool_context_t context, worker_pool_job_t * job, unsigned int n);
-
-WORKERPOOL_API AEEResult worker_pool_run_func(worker_pool_context_t context,
- worker_callback_t func,
- void * data,
- unsigned int n);
-
-WORKERPOOL_API AEEResult worker_pool_set_thread_priority(worker_pool_context_t context, unsigned int prio);
-WORKERPOOL_API AEEResult worker_pool_get_thread_priority(worker_pool_context_t context, unsigned int * prio);
-WORKERPOOL_API AEEResult worker_pool_retrieve_thread_id(worker_pool_context_t context, unsigned int * tids);
-
-#ifdef __cplusplus
-}
-#endif
-
-#endif // #ifndef HTP_WORKER_POOL_H
if args.timeline:
logger.info(f"\n# ASCII Timing {args.timeline.capitalize()}\n")
- printed_cnt = 0
for op in ops:
if args.timeline == "summary":
print_ascii_summary(op['name'], op['dims'], op['types'], op['usec'], op['cycles'], op['trace_events'], op.get('evt_val'))
elif args.timeline == "diagram":
print_ascii_timeline(op['name'], op['dims'], op['types'], op['usec'], op['cycles'], op['trace_events'], op.get('evt_val'))
- printed_cnt += 1
- if printed_cnt >= args.top:
- break
else:
generate_report(ops, args.top, overrides, args.sort, pmu_name=final_pmu_name)