Edge AI Add-on API 2.3.0
Loading...
Searching...
No Matches
nrf_axon_nn_infer.h
Go to the documentation of this file.
1/*
2 * Copyright (c) 2025 Nordic Semiconductor ASA
3 *
4 * SPDX-License-Identifier: LicenseRef-Nordic-5-Clause
5 */
6
7#pragma once
8
9#ifdef __cplusplus
10extern "C" {
11#endif
12#include <stdarg.h>
13#include <stdint.h>
14#include "nrf_axon_driver.h"
15
23#ifndef NRF_AXON_MODEL_APP_STORAGE
24# define NRF_AXON_MODEL_APP_STORAGE
25#endif
29typedef struct {
30 uint16_t height;
31 uint16_t width;
32 uint16_t channel_cnt;
33 int16_t batch_cnt;
34 uint8_t byte_width;
36
62
68typedef struct {
70 int8_t *buf_ptr;
72 uint32_t buf_size;
76 uint8_t byte_width;
78
90 int8_t *ptr;
91 /* offset into the packed buffer of this output */
93 /* packed size of this output */
94 uint32_t packed_size;
102 /* dequantization multiplier. */
103 uint32_t dequant_mult;
104 /* node id of the output */
105 int16_t node_id;
110 /*
111 * length in bytes of the distance between the start of rows in the unpacked
112 * output in the interlayer byffer.
113 */
114 uint16_t stride;
116
122 /*
123 * version of the compiler that generated the model. bits 23:16 => major,
124 * bits 15:8 => minor, bits 8:0 => patch
125 */
127 /* name of the model provided by user at compilation time.*/
128 const char *model_name;
129 /*
130 * optional list of text labels that correspond to classification indices.
131 * Applies to single dimension classification models only.
132 */
133 const char **labels;
136 /*
137 * list of ptrs to model input vectors that is sized to input_cnt that
138 * is provisioned by the model.
139 * Each index is the input vector for the corresponding entry in inputs[].
140 * User is required to populate this list before calling inference APIs.
141 */
142 const int8_t **input_vector_list;
144 int16_t input_cnt;
156 const void *model_const_ptr;
164 struct {
166 int8_t *buf_ptr;
168 uint32_t buf_size;
172 uint16_t count;
174 /* number of model outputs pointed to by outputs */
175 uint16_t output_cnt;
176 /* pointer to the outputs of the model */
182 /* model requires this version of the driver or later to execute properly. */
184 /*
185 * layer model is a superset of the full model. If true, this can be treated as a
186 * nrf_axon_nn_compiled_model_layer_s.
187 */
190
199static inline void nrf_axon_nn_set_input_vector(const nrf_axon_nn_compiled_model_s *compiled_model,
200 uint16_t input_no, const int8_t *input_vector)
201{
202 if (input_no < compiled_model->input_cnt) {
203 compiled_model->input_vector_list[input_no] = input_vector;
204 }
205}
206
222 const nrf_axon_nn_compiled_model_s *compiled_model);
223
224
235 const nrf_axon_nn_compiled_model_s *compiled_model);
236
268 const nrf_axon_nn_compiled_model_s *compiled_model,
269 const int8_t *input_vector,
270 int8_t *output_buffer);
271
306 const nrf_axon_nn_compiled_model_s *compiled_model,
307 int8_t *output_buffer);
308
322
328typedef struct {
336 void (*inference_callback)(nrf_axon_result_e result, void *callback_context);
339 /*
340 * populated and managed by the driver. dedicated buffers outside the interlayer buffer
341 * for storing model outputs.
342 */
349
364 const nrf_axon_nn_compiled_model_s *compiled_model);
365
374
375/*
376 * @brief Starts an asynchronous inference on the provided model.
377
378 * In asynchonous mode, models are inferred in a separate thread, one after another.
379 * User provides input vector and output buffer information as the interlayer buffer
380 * is used by all models, so only when it is this model's turn to execute can its
381 * input be populated from the input_vector.
382 *
383 * Upon completion of inference the next job is queued before the user callback is invoked,
384 * so the results have to be copied by the driver to the output_buffer before invoking the
385 * user callback.
386 *
387 * @param[in] model_wrapper Model to run inference on, initialized via a one-time call to
388 * nrf_axon_nn_model_async_init.
389 * @param[in] input_vector Input to run inference on. It is not consumed immediately so has to
390 * be in memory that is valid as long as inference is occurring.
391 * @param[in] output_buffer buffer to copy inference results to.
392 * @param[in] inference_callback Function to invoke when inference has completed.
393 * @param[in] callback_context Opaque pointer provided to inference_callback.
394 * @retval[0] Inference successfully queued.
395 * @retval[NRF_AXON_RESULT_NOT_FINISHED] Model is still busy with an ealier inference
396 * @retval[<0] Error code.
397 */
400 const int8_t *input_vector,
401 int8_t *output_buffer,
402 void (*inference_callback)(nrf_axon_result_e result, void *callback_context),
403 void *callback_context);
404
407 const int8_t **input_vector_list,
408 int8_t *output_buffer,
409 void (*inference_callback)(nrf_axon_result_e result, void *callback_context),
410 void *callback_context);
427 const nrf_axon_nn_compiled_model_s *compiled_model,
428 const int8_t *packed_output,
429 const char **label, int32_t *score);
430
449 const nrf_axon_nn_compiled_model_s *compiled_model);
450
465 const nrf_axon_nn_compiled_model_s *compiled_model,
466 uint8_t output_ndx);
467
483 const nrf_axon_nn_compiled_model_s *compiled_model,
484 int8_t *to_buffer);
485
486#ifdef __cplusplus
487} /* extern "C" { */
488#endif
nrf_axon_result_e
Axon driver return codes.
Definition nrf_axon_driver.h:131
uint32_t NRF_AXON_PLATFORM_BITWIDTH_UNSIGNED_TYPE
Definition nrf_axon_driver.h:105
nrf_axon_result_e nrf_axon_nn_model_infer_sync_multi_inputs(const nrf_axon_nn_compiled_model_s *compiled_model, int8_t *output_buffer)
Blocking inference function of a compiled model with multiple inputs.
nrf_axon_result_e nrf_axon_nn_model_infer_async_multi_inputs(nrf_axon_nn_model_async_inference_wrapper_s *model_wrapper, const int8_t **input_vector_list, int8_t *output_buffer, void(*inference_callback)(nrf_axon_result_e result, void *callback_context), void *callback_context)
void nrf_axon_nn_copy_output_to_packed_buffer(const nrf_axon_nn_compiled_model_s *compiled_model, int8_t *to_buffer)
Copies and packs the model inference output from the common interlayer buffer to the user's dedicated...
int nrf_axon_nn_offset_to_output_ndx(const nrf_axon_nn_compiled_model_s *compiled_model, uint8_t output_ndx)
Returns the offset into the packed output buffer filled for the start of a particular output node ind...
struct nrf_axon_compiled_model_output_tag_s nrf_axon_compiled_model_output_s
nrf_axon_result_e nrf_axon_nn_model_infer_sync(const nrf_axon_nn_compiled_model_s *compiled_model, const int8_t *input_vector, int8_t *output_buffer)
Blocking inference function of a compiled model that has only one input.
nrf_axon_result_e nrf_axon_nn_model_async_init(nrf_axon_nn_model_async_inference_wrapper_s *model_wrapper, const nrf_axon_nn_compiled_model_s *compiled_model)
Initialize a model for asynchronous inference. Calls nrf_axon_nn_model_validate then binds the model ...
int nrf_axon_nn_model_init_vars(const nrf_axon_nn_compiled_model_s *compiled_model)
Initialize all the persistent var buffers in a streaming-style model (with VarHandle/ReadVariable/Ass...
struct nrf_axon_nn_compiled_model_tag_s nrf_axon_nn_compiled_model_s
nrf_axon_nn_async_inference_status_e
Asynchronous inference states.
Definition nrf_axon_nn_infer.h:314
@ NRF_AXON_NN_ASYNC_INFERENCE_STATUS_IDLE
Definition nrf_axon_nn_infer.h:316
@ NRF_AXON_NN_ASYNC_INFERENCE_STATUS_COMPLETE
Definition nrf_axon_nn_infer.h:320
@ NRF_AXON_NN_ASYNC_INFERENCE_STATUS_ACTIVE
Definition nrf_axon_nn_infer.h:318
nrf_axon_result_e nrf_axon_nn_model_infer_async(nrf_axon_nn_model_async_inference_wrapper_s *model_wrapper, const int8_t *input_vector, int8_t *output_buffer, void(*inference_callback)(nrf_axon_result_e result, void *callback_context), void *callback_context)
static void nrf_axon_nn_set_input_vector(const nrf_axon_nn_compiled_model_s *compiled_model, uint16_t input_no, const int8_t *input_vector)
Populates input_vector_list[input_no] with input_vector.
Definition nrf_axon_nn_infer.h:199
nrf_axon_nn_async_inference_status_e nrf_axon_nn_get_model_async_infer_status(const nrf_axon_nn_model_async_inference_wrapper_s *model_wrapper)
Returns the model inference status of an asynchronous inference.
struct nrf_axon_nn_compiled_model_input_tag_s nrf_axon_nn_compiled_model_input_s
int16_t nrf_axon_nn_get_classification(const nrf_axon_nn_compiled_model_s *compiled_model, const int8_t *packed_output, const char **label, int32_t *score)
Gets the inference results for a classification model.
nrf_axon_result_e nrf_axon_nn_populate_input_vectors(const nrf_axon_nn_compiled_model_s *compiled_model)
Copies model input from input_vector to the location in the interlayer buffer the model expects.
nrf_axon_result_e nrf_axon_nn_model_validate(const nrf_axon_nn_compiled_model_s *compiled_model)
Sanity check of a compiled model.
Internal structure supplied by user but managed by the driver to track execution progress.
Definition nrf_axon_driver.h:173
uint32_t packed_buffer_offset
Definition nrf_axon_nn_infer.h:92
int8_t * ptr
Definition nrf_axon_nn_infer.h:90
uint16_t stride
Definition nrf_axon_nn_infer.h:114
uint32_t packed_size
Definition nrf_axon_nn_infer.h:94
int16_t node_id
Definition nrf_axon_nn_infer.h:105
nrf_axon_nn_model_layer_dimensions_s dimensions
Definition nrf_axon_nn_infer.h:95
uint32_t dequant_mult
Definition nrf_axon_nn_infer.h:103
uint8_t dequant_round
Definition nrf_axon_nn_infer.h:107
int8_t dequant_zp
Definition nrf_axon_nn_infer.h:109
Definition nrf_axon_nn_infer.h:88
uint16_t stride
Definition nrf_axon_nn_infer.h:56
nrf_axon_nn_model_layer_dimensions_s dimensions
Definition nrf_axon_nn_infer.h:44
int16_t node_id
Definition nrf_axon_nn_infer.h:54
int8_t quant_zp
Definition nrf_axon_nn_infer.h:60
int8_t * ptr
Definition nrf_axon_nn_infer.h:42
uint8_t quant_round
Definition nrf_axon_nn_infer.h:58
uint32_t quant_mult
Definition nrf_axon_nn_infer.h:49
Definition nrf_axon_nn_infer.h:40
int8_t * buf_ptr
Definition nrf_axon_nn_infer.h:166
struct nrf_axon_nn_compiled_model_tag_s::@3 persistent_vars
const nrf_axon_nn_model_persistent_var_s * vars
Definition nrf_axon_nn_infer.h:170
uint32_t model_const_size
Definition nrf_axon_nn_infer.h:158
const NRF_AXON_PLATFORM_BITWIDTH_UNSIGNED_TYPE * cmd_buffer_ptr
Definition nrf_axon_nn_infer.h:154
uint16_t output_cnt
Definition nrf_axon_nn_infer.h:175
uint32_t interlayer_buffer_needed
Definition nrf_axon_nn_infer.h:148
uint32_t psum_buffer_needed
Definition nrf_axon_nn_infer.h:152
int8_t * packed_output_buf
Definition nrf_axon_nn_infer.h:181
bool is_layer_model
Definition nrf_axon_nn_infer.h:188
int16_t input_cnt
Definition nrf_axon_nn_infer.h:144
uint32_t compiler_version
Definition nrf_axon_nn_infer.h:126
uint32_t buf_size
Definition nrf_axon_nn_infer.h:168
const nrf_axon_nn_compiled_model_input_s * inputs
Definition nrf_axon_nn_infer.h:135
const int8_t ** input_vector_list
Definition nrf_axon_nn_infer.h:142
uint16_t count
Definition nrf_axon_nn_infer.h:172
const nrf_axon_compiled_model_output_s * outputs
Definition nrf_axon_nn_infer.h:177
uint32_t min_driver_version_required
Definition nrf_axon_nn_infer.h:183
const char * model_name
Definition nrf_axon_nn_infer.h:128
uint32_t cmd_buffer_len
Definition nrf_axon_nn_infer.h:160
const char ** labels
Definition nrf_axon_nn_infer.h:133
const void * model_const_ptr
Definition nrf_axon_nn_infer.h:156
Definition nrf_axon_nn_infer.h:121
const nrf_axon_nn_compiled_model_s * compiled_model
Definition nrf_axon_nn_infer.h:330
nrf_axon_nn_async_inference_status_e infer_status
Definition nrf_axon_nn_infer.h:347
nrf_axon_queued_cmd_info_wrapper_s queued_cmd_buf_wrapper
Definition nrf_axon_nn_infer.h:334
void * callback_context
Definition nrf_axon_nn_infer.h:338
nrf_axon_cmd_buffer_info_s cmd_buf_info
Definition nrf_axon_nn_infer.h:332
int8_t * output_buffer
Definition nrf_axon_nn_infer.h:343
Combines the compiled model info with other structures to support asynchronous inferencing....
Definition nrf_axon_nn_infer.h:328
int16_t batch_cnt
Definition nrf_axon_nn_infer.h:33
uint16_t width
Definition nrf_axon_nn_infer.h:31
uint16_t height
Definition nrf_axon_nn_infer.h:30
uint16_t channel_cnt
Definition nrf_axon_nn_infer.h:32
uint8_t byte_width
Definition nrf_axon_nn_infer.h:34
Definition nrf_axon_nn_infer.h:29
int32_t initial_value
Definition nrf_axon_nn_infer.h:74
int8_t * buf_ptr
Definition nrf_axon_nn_infer.h:70
uint32_t buf_size
Definition nrf_axon_nn_infer.h:72
uint8_t byte_width
Definition nrf_axon_nn_infer.h:76
Definition nrf_axon_nn_infer.h:68
Used as a parameter to nrf_axon_queue_cmd_buf for asynchronous execution.
Definition nrf_axon_driver.h:225