{
  "$schema": "https://ambiqai.github.io/helia-ui/schema/reference-model-1.json",
  "generatedFrom": {
    "sourceCommit": "5f3fed9f21a57390cc7f00f77a37db8f5f110cb8",
    "tool": "doxyref",
    "version": "1.17.0"
  },
  "language": "c",
  "modules": [
    {
      "description": "Internal Support functions. Not intended to be called direclty by a CMSIS-NN user.",
      "name": "Private",
      "path": "heliaCORE.groupSupport",
      "submodules": [],
      "summary": "Internal Support functions.",
      "symbols": [
        {
          "description": "Polynomial coefficients used by the float32 MVE exp approximation.",
          "examples": [],
          "id": "arm_nn_exp_poly_coeffs_f32",
          "kind": "attribute",
          "language": "c",
          "members": [],
          "name": "arm_nn_exp_poly_coeffs_f32",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "const float32_t arm_nn_exp_poly_coeffs_f32[8]",
          "source": {
            "line": 70,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L70"
          },
          "summary": "Polynomial coefficients used by the float32 MVE exp approximation."
        },
        {
          "description": "LUT for `2^(i/256)` used by the float32 LUT softmax approximation.\n\nStores 257 samples for `i = 0..256` so interpolation can safely read `lut[idx + 1]` while indexing the 256 fractional segments.",
          "examples": [],
          "id": "arm_nn_exp2_lut_f32",
          "kind": "attribute",
          "language": "c",
          "members": [],
          "name": "arm_nn_exp2_lut_f32",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "const float32_t arm_nn_exp2_lut_f32[257]",
          "source": {
            "line": 78,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L78"
          },
          "summary": "LUT for 2^(i/256) used by the float32 LUT softmax approximation."
        },
        {
          "description": "Floor of `x` as an int32_t.\n\nPrecondition: `x` must already be reduced to the int32_t range and must not be NaN  the float-to-int conversion below is undefined otherwise. The only caller, arm_nn_softmax_exp_lut_f32(), guarantees this by clamping its input to [-80, 80] (NaN included, see there) before scaling by log2(e), which bounds `x` to +/-116.",
          "examples": [],
          "id": "arm_nn_softmax_floor_to_int_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_floor_to_int_f32",
          "params": [
            {
              "description": "Value to floor.",
              "direction": "in",
              "name": "x",
              "type": "float32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Largest int32_t not greater than `x`."
            }
          ],
          "signature": "static int32_t arm_nn_softmax_floor_to_int_f32(float32_t x)",
          "source": {
            "line": 92,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L92"
          },
          "summary": "Floor of x as an int32t."
        },
        {
          "description": "Reinterpret a 32-bit pattern as a float32.",
          "examples": [],
          "id": "arm_nn_softmax_fp32_from_bits",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_fp32_from_bits",
          "params": [
            {
              "description": "IEEE-754 binary32 bit pattern.",
              "direction": "in",
              "name": "bits",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The float32 value with the bit pattern `bits`."
            }
          ],
          "signature": "static float32_t arm_nn_softmax_fp32_from_bits(uint32_t bits)",
          "source": {
            "line": 104,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L104"
          },
          "summary": "Reinterpret a 32-bit pattern as a float32."
        },
        {
          "description": "Compute `2^n` as a float32 by building the exponent field directly.",
          "examples": [],
          "id": "arm_nn_softmax_exp2i_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_exp2i_f32",
          "params": [
            {
              "description": "Integer exponent. Clamped to the normal float32 exponent range `[-126, 127]`.",
              "direction": "in",
              "name": "n",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`2^n` as a float32."
            }
          ],
          "signature": "static float32_t arm_nn_softmax_exp2i_f32(int32_t n)",
          "source": {
            "line": 121,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L121"
          },
          "summary": "Compute 2^n as a float32 by building the exponent field directly."
        },
        {
          "description": "Taylor/Estrin exp approximation for float32 softmax helpers.\n\nThe polynomial is evaluated on r in [-ln2/2, ln2/2]. Coefficients come from the Maclaurin series of exp(r): exp(r) ~= 1 + r + r^2/2! + r^3/3! + r^4/4! + r^5/5! + r^6/6! Grouped via Estrin to reduce dependency depth: p = (1 + r) + r^2*(1/2 + r/6) + r^4*(1/24 + r/120) + r^6*(1/720)\n\nRange reduction follows: x = n * ln(2) + r, exp(x) = exp(r) * 2^n",
          "examples": [],
          "id": "arm_nn_softmax_exp_taylor_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_exp_taylor_f32",
          "params": [
            {
              "description": "Exponent argument. Clamped to `[-80, 80]` before evaluation.",
              "direction": "in",
              "name": "x",
              "type": "float32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Approximation of `exp(x)`."
            }
          ],
          "signature": "static float32_t arm_nn_softmax_exp_taylor_f32(float32_t x)",
          "source": {
            "line": 147,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L147"
          },
          "summary": "Taylor/Estrin exp approximation for float32 softmax helpers."
        },
        {
          "description": "LUT-based exp approximation for float32 softmax helpers.\n\nSplits `x * log2(e)` into an integer part handled by arm_nn_softmax_exp2i_f32() and a fractional part interpolated linearly from `arm_nn_exp2_lut_f32`.",
          "examples": [],
          "id": "arm_nn_softmax_exp_lut_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_exp_lut_f32",
          "params": [
            {
              "description": "Exponent argument. Clamped to `[-80, 80]` before evaluation; NaN is flushed to `80`.",
              "direction": "in",
              "name": "x",
              "type": "float32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Approximation of `exp(x)`."
            }
          ],
          "signature": "static float32_t arm_nn_softmax_exp_lut_f32(float32_t x)",
          "source": {
            "line": 184,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L184"
          },
          "summary": "LUT-based exp approximation for float32 softmax helpers."
        },
        {
          "description": "Scalar exp approximation used by the float32 softmax paths.\n\nDispatches to arm_nn_softmax_exp_taylor_f32() when `ARM_NN_USE_EXP_TAYLOR` is defined and to arm_nn_softmax_exp_lut_f32() otherwise.",
          "examples": [],
          "id": "arm_nn_softmax_exp_scalar_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_exp_scalar_f32",
          "params": [
            {
              "description": "Exponent argument.",
              "direction": "in",
              "name": "x",
              "type": "float32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Approximation of `exp(x)`."
            }
          ],
          "signature": "static float32_t arm_nn_softmax_exp_scalar_f32(float32_t x)",
          "source": {
            "line": 254,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L254"
          },
          "summary": "Scalar exp approximation used by the float32 softmax paths."
        },
        {
          "description": "LUT for tanh(x) sampled over `x in [0, 6]` for float32 helpers.\n\nStores 385 samples so interpolation can safely read `lut[idx + 1]` while indexing the 384 fractional segments across the interval. The grid spacing (`6/384 == 1/64`) matches the earlier 257-entry `[0, 4]` table, so entries `0..256` are bit-identical to it and the index multiplier is unchanged. Generated by `scripts/gen_tanh_lut_f32.py`.",
          "examples": [],
          "id": "arm_nn_tanh_lut_f32",
          "kind": "attribute",
          "language": "c",
          "members": [],
          "name": "arm_nn_tanh_lut_f32",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "const float32_t arm_nn_tanh_lut_f32[385]",
          "source": {
            "line": 276,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L276"
          },
          "summary": "LUT for tanh(x) sampled over x in [0, 6] for float32 helpers."
        },
        {
          "description": "Copy a float32 vector.",
          "examples": [],
          "id": "arm_memcpy_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memcpy_f32",
          "params": [
            {
              "description": "Destination buffer.",
              "direction": "out",
              "name": "dst",
              "type": "float32_t *"
            },
            {
              "description": "Source buffer.",
              "direction": "in",
              "name": "src",
              "type": "const float32_t *"
            },
            {
              "description": "Number of elements to copy.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memcpy_f32(float32_t *dst, const float32_t *src, uint32_t block_size)",
          "source": {
            "line": 311,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L311"
          },
          "summary": "Copy a float32 vector."
        },
        {
          "description": "Set a float32 vector to a constant value.",
          "examples": [],
          "id": "arm_memset_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memset_f32",
          "params": [
            {
              "description": "Destination buffer.",
              "direction": "out",
              "name": "dst",
              "type": "float32_t *"
            },
            {
              "description": "Fill value.",
              "direction": "in",
              "name": "val",
              "type": "const float32_t"
            },
            {
              "description": "Number of elements to write.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memset_f32(float32_t *dst, const float32_t val, uint32_t block_size)",
          "source": {
            "line": 334,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L334"
          },
          "summary": "Set a float32 vector to a constant value."
        },
        {
          "description": "Specialized NHWC depthwise 1D kernel for `k=3`, `ch_mult=1` (float32).",
          "examples": [],
          "id": "arm_nn_depthwise_conv1d_k3_nhwc_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv1d_k3_nhwc_f32",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float32_t *"
            },
            {
              "description": "Number of input (and output) channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Depthwise weights with shape `[3][in_c]`.",
              "direction": "in",
              "name": "kernel",
              "type": "const float32_t *"
            },
            {
              "description": "Optional bias vector of `in_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float32_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][in_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float32_t *"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_depthwise_conv1d_k3_nhwc_f32(\n    const float32_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float32_t *kernel,\n    const float32_t *b,\n    float32_t *out,\n    int32_t out_w\n)",
          "source": {
            "line": 367,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L367"
          },
          "summary": "Specialized NHWC depthwise 1D kernel for k=3, chmult=1 (float32)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=5` (float32).",
          "examples": [],
          "id": "arm_nn_conv1d_k5_nhwc_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k5_nhwc_f32",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float32_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights with shape `[out_c][5][in_c]`.",
              "direction": "in",
              "name": "kernel",
              "type": "const float32_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float32_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float32_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+4`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k5_nhwc_f32(\n    const float32_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float32_t *kernel,\n    const float32_t *b,\n    float32_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 387,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L387"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=5 (float32)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=5` (float32, packed weights).\n\nThe packed kernel uses the same `NTxN` RHS layout as `arm_nn_mat_mult_nt_n_packed_f32`, i.e. `[(5 * in_c)][out_c_block_of_4]`.",
          "examples": [],
          "id": "arm_nn_conv1d_k5_packed_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k5_packed_f32",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float32_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights packed in output-channel blocks of 4 as described above.",
              "direction": "in",
              "name": "kernel_packed",
              "type": "const float32_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float32_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float32_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+4`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k5_packed_f32(\n    const float32_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float32_t *kernel_packed,\n    const float32_t *b,\n    float32_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 411,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L411"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=5 (float32, packed weights)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=3` (float32).",
          "examples": [],
          "id": "arm_nn_conv1d_k3_nhwc_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k3_nhwc_f32",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float32_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights with shape `[out_c][3][in_c]`.",
              "direction": "in",
              "name": "kernel",
              "type": "const float32_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float32_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float32_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k3_nhwc_f32(\n    const float32_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float32_t *kernel,\n    const float32_t *b,\n    float32_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 432,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L432"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=3 (float32)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=3` (float32, packed weights).\n\nThe packed kernel uses the same `NTxN` RHS layout as `arm_nn_mat_mult_nt_n_packed_f32`, i.e. `[(3 * in_c)][out_c_block_of_4]`.",
          "examples": [],
          "id": "arm_nn_conv1d_k3_packed_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k3_packed_f32",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float32_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights packed in output-channel blocks of 4 as described above.",
              "direction": "in",
              "name": "kernel_packed",
              "type": "const float32_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float32_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float32_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k3_packed_f32(\n    const float32_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float32_t *kernel_packed,\n    const float32_t *b,\n    float32_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 456,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L456"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=3 (float32, packed weights)."
        },
        {
          "description": "Specialized NHWC max-pool 1D kernel for `k=3`, `s=3` (float32).",
          "examples": [],
          "id": "arm_nn_maxpool1d_k3s3_nhwc_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_maxpool1d_k3s3_nhwc_f32",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float32_t *"
            },
            {
              "description": "Number of channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][in_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float32_t *"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `3*ow..3*ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_maxpool1d_k3s3_nhwc_f32(const float32_t *x_nhwc, int32_t in_c, int32_t in_w, float32_t *out, int32_t out_w)",
          "source": {
            "line": 474,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L474"
          },
          "summary": "Specialized NHWC max-pool 1D kernel for k=3, s=3 (float32)."
        },
        {
          "description": "Specialized NHWC max-pool 1D kernel for `k=2`, `s=2` without output clamp (float32).",
          "examples": [],
          "id": "arm_nn_maxpool1d_k2s2_nhwc_noclip_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_maxpool1d_k2s2_nhwc_noclip_f32",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float32_t *"
            },
            {
              "description": "Number of channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][in_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float32_t *"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `2*ow..2*ow+1`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_maxpool1d_k2s2_nhwc_noclip_f32(\n    const float32_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    float32_t *out,\n    int32_t out_w\n)",
          "source": {
            "line": 489,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L489"
          },
          "summary": "Specialized NHWC max-pool 1D kernel for k=2, s=2 without output clamp (float32)."
        },
        {
          "description": "Specialized NHWC max-pool 1D kernel for `k=2`, `s=2` with clamp (float32).",
          "examples": [],
          "id": "arm_nn_maxpool1d_k2s2_nhwc_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_maxpool1d_k2s2_nhwc_f32",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float32_t *"
            },
            {
              "description": "Number of channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][in_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float32_t *"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `2*ow..2*ow+1`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            },
            {
              "description": "Lower clamp bound applied to `out`.",
              "direction": "in",
              "name": "act_min",
              "type": "float32_t"
            },
            {
              "description": "Upper clamp bound applied to `out`.",
              "direction": "in",
              "name": "act_max",
              "type": "float32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_maxpool1d_k2s2_nhwc_f32(\n    const float32_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    float32_t *out,\n    int32_t out_w,\n    float32_t act_min,\n    float32_t act_max\n)",
          "source": {
            "line": 506,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L506"
          },
          "summary": "Specialized NHWC max-pool 1D kernel for k=2, s=2 with clamp (float32)."
        },
        {
          "description": "Matrix multiply with non-transposed lhs and transposed rhs rows (float32).",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_t_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_t_f32",
          "params": [
            {
              "description": "Left-hand matrix stored row-major.",
              "direction": "in",
              "name": "lhs",
              "type": "const float32_t *"
            },
            {
              "description": "Right-hand matrix stored row-major, one row per output channel.",
              "direction": "in",
              "name": "rhs",
              "type": "const float32_t *"
            },
            {
              "description": "Optional bias vector.",
              "direction": "in",
              "name": "bias",
              "type": "const float32_t *"
            },
            {
              "description": "Output matrix.",
              "direction": "out",
              "name": "dst",
              "type": "float32_t *"
            },
            {
              "description": "Number of rows in `lhs`.",
              "direction": "in",
              "name": "lhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Number of rows in `rhs`.",
              "direction": "in",
              "name": "rhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Number of columns in `rhs`.",
              "direction": "in",
              "name": "rhs_cols",
              "type": "int32_t"
            },
            {
              "description": "Output row stride, expressed in elements.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "int32_t"
            },
            {
              "description": "Lower clamp bound.",
              "direction": "in",
              "name": "activation_min",
              "type": "float32_t"
            },
            {
              "description": "Upper clamp bound.",
              "direction": "in",
              "name": "activation_max",
              "type": "float32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`ARM_CMSIS_NN_SUCCESS` on success or `ARM_CMSIS_NN_ARG_ERROR` on invalid arguments."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_t_f32(\n    const float32_t *lhs,\n    const float32_t *rhs,\n    const float32_t *bias,\n    float32_t *dst,\n    int32_t lhs_rows,\n    int32_t rhs_rows,\n    int32_t rhs_cols,\n    int32_t row_address_offset,\n    float32_t activation_min,\n    float32_t activation_max\n)",
          "source": {
            "line": 529,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L529"
          },
          "summary": "Matrix multiply with non-transposed lhs and transposed rhs rows (float32)."
        },
        {
          "description": "Matrix multiply with non-transposed lhs and packed non-transposed rhs (float32).",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_n_packed_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_n_packed_f32",
          "params": [
            {
              "description": "Left-hand matrix stored row-major with logical shape `[lhs_rows, rhs_cols]`.",
              "direction": "in",
              "name": "lhs",
              "type": "const float32_t *"
            },
            {
              "description": "Right-hand matrix with logical shape `[rhs_cols, rhs_rows]`, packed in column blocks of 4. The final block uses the same packed stride and inactive tail lanes are ignored.",
              "direction": "in",
              "name": "rhs_packed",
              "type": "const float32_t *"
            },
            {
              "description": "Optional bias vector.",
              "direction": "in",
              "name": "bias",
              "type": "const float32_t *"
            },
            {
              "description": "Output matrix.",
              "direction": "out",
              "name": "dst",
              "type": "float32_t *"
            },
            {
              "description": "Number of rows in `lhs`.",
              "direction": "in",
              "name": "lhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Number of logical output columns in the unpacked rhs matrix.",
              "direction": "in",
              "name": "rhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Shared reduction dimension `K`.",
              "direction": "in",
              "name": "rhs_cols",
              "type": "int32_t"
            },
            {
              "description": "Output row stride, expressed in elements.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "int32_t"
            },
            {
              "description": "Lower clamp bound.",
              "direction": "in",
              "name": "activation_min",
              "type": "float32_t"
            },
            {
              "description": "Upper clamp bound.",
              "direction": "in",
              "name": "activation_max",
              "type": "float32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`ARM_CMSIS_NN_SUCCESS` on success or `ARM_CMSIS_NN_ARG_ERROR` on invalid arguments."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_n_packed_f32(\n    const float32_t *lhs,\n    const float32_t *rhs_packed,\n    const float32_t *bias,\n    float32_t *dst,\n    int32_t lhs_rows,\n    int32_t rhs_rows,\n    int32_t rhs_cols,\n    int32_t row_address_offset,\n    float32_t activation_min,\n    float32_t activation_max\n)",
          "source": {
            "line": 556,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L556"
          },
          "summary": "Matrix multiply with non-transposed lhs and packed non-transposed rhs (float32)."
        },
        {
          "description": "Pack a single convolution patch into one row of a contiguous float32 patch matrix.\n\nDevelopers familiar with im2row/im2col terminology can think of this as packing one output patch into one row.",
          "examples": [],
          "id": "arm_nn_pack_conv_patch_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_pack_conv_patch_f32",
          "params": [
            {
              "description": "Input tensor for one batch in NHWC layout with shape `[in_h][in_w][in_c]`.",
              "direction": "in",
              "name": "input",
              "type": "const float32_t *"
            },
            {
              "description": "Input height.",
              "direction": "in",
              "name": "in_h",
              "type": "int32_t"
            },
            {
              "description": "Input width.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Kernel height.",
              "direction": "in",
              "name": "kernel_h",
              "type": "int32_t"
            },
            {
              "description": "Kernel width.",
              "direction": "in",
              "name": "kernel_w",
              "type": "int32_t"
            },
            {
              "description": "Vertical stride.",
              "direction": "in",
              "name": "stride_h",
              "type": "int32_t"
            },
            {
              "description": "Horizontal stride.",
              "direction": "in",
              "name": "stride_w",
              "type": "int32_t"
            },
            {
              "description": "Top padding.",
              "direction": "in",
              "name": "pad_h",
              "type": "int32_t"
            },
            {
              "description": "Left padding.",
              "direction": "in",
              "name": "pad_w",
              "type": "int32_t"
            },
            {
              "description": "Vertical dilation.",
              "direction": "in",
              "name": "dilation_h",
              "type": "int32_t"
            },
            {
              "description": "Horizontal dilation.",
              "direction": "in",
              "name": "dilation_w",
              "type": "int32_t"
            },
            {
              "description": "Output row index of the patch to pack.",
              "direction": "in",
              "name": "out_y",
              "type": "int32_t"
            },
            {
              "description": "Output column index of the patch to pack.",
              "direction": "in",
              "name": "out_x",
              "type": "int32_t"
            },
            {
              "description": "Value written for taps that fall outside the input.",
              "direction": "in",
              "name": "pad_value",
              "type": "float32_t"
            },
            {
              "description": "Destination row of `kernel_h * kernel_w * in_c` elements, ordered `[kernel_h][kernel_w][in_c]`.",
              "direction": "out",
              "name": "patch_row",
              "type": "float32_t *"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_pack_conv_patch_f32(\n    const float32_t *input,\n    int32_t in_h,\n    int32_t in_w,\n    int32_t in_c,\n    int32_t kernel_h,\n    int32_t kernel_w,\n    int32_t stride_h,\n    int32_t stride_w,\n    int32_t pad_h,\n    int32_t pad_w,\n    int32_t dilation_h,\n    int32_t dilation_w,\n    int32_t out_y,\n    int32_t out_x,\n    float32_t pad_value,\n    float32_t *patch_row\n)",
          "source": {
            "line": 590,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L590"
          },
          "summary": "Pack a single convolution patch into one row of a contiguous float32 patch matrix."
        },
        {
          "description": "Specialized softmax helper for a single float32 row of length 2.",
          "examples": [],
          "id": "arm_nn_softmax_1x2_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_1x2_f32",
          "params": [
            {
              "description": "Pointer to two contiguous float32 input values.",
              "direction": "in",
              "name": "in",
              "type": "const float32_t *"
            },
            {
              "description": "Pointer to two contiguous float32 output values.",
              "direction": "out",
              "name": "out",
              "type": "float32_t *"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_softmax_1x2_f32(const float32_t *in, float32_t *out)",
          "source": {
            "line": 613,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L613"
          },
          "summary": "Specialized softmax helper for a single float32 row of length 2."
        },
        {
          "description": "Blockwise float16 accumulation on the MVE legs (AmbiqAI/ns-cmsis-nn#586).\n\nA float16 accumulator lane sums at most ARM_NN_F16_ACC_BLOCK taps, in the kernel's tap order, before its partial is widened exactly and added into a float32 accumulator; the float32 sum rounds to float16 once. The `_acc16` entries instantiate the same kernel bodies with ARM_NN_F16_ACC_BLOCK_NONE, which never folds.",
          "examples": [],
          "id": "ARM_NN_F16_ACC_BLOCK",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_F16_ACC_BLOCK",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_F16_ACC_BLOCK (32)",
          "source": {
            "line": 626,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L626"
          },
          "summary": "Blockwise float16 accumulation on the MVE legs (AmbiqAI/ns-cmsis-nn#586)."
        },
        {
          "description": "",
          "examples": [],
          "id": "ARM_NN_F16_ACC_BLOCK_NONE",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_F16_ACC_BLOCK_NONE",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_F16_ACC_BLOCK_NONE (INT32_MAX)",
          "source": {
            "line": 627,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L627"
          },
          "summary": ""
        },
        {
          "description": "Polynomial coefficients used by the float16 MVE exp approximation.\n\nThe float16 MVE helper evaluates the polynomial in widened float32 lanes, but it uses a dedicated coefficient table to keep the float16 path isolated from the float32 feature gate and softmax support stack.",
          "examples": [],
          "id": "arm_nn_exp_poly_coeffs_f16",
          "kind": "attribute",
          "language": "c",
          "members": [],
          "name": "arm_nn_exp_poly_coeffs_f16",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "const float32_t arm_nn_exp_poly_coeffs_f16[8]",
          "source": {
            "line": 636,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L636"
          },
          "summary": "Polynomial coefficients used by the float16 MVE exp approximation."
        },
        {
          "description": "Quantized binary16 LUT for `2^(i/256)` used by float16 helpers.\n\nStores 257 samples for `i = 0..256` so interpolation can safely read `lut[idx + 1]` while indexing the 256 fractional segments.",
          "examples": [],
          "id": "arm_nn_exp2_lut_f16",
          "kind": "attribute",
          "language": "c",
          "members": [],
          "name": "arm_nn_exp2_lut_f16",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "const uint16_t arm_nn_exp2_lut_f16[257]",
          "source": {
            "line": 644,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L644"
          },
          "summary": "Quantized binary16 LUT for 2^(i/256) used by float16 helpers."
        },
        {
          "description": "Quantized binary16 LUT for tanh(x) with `x in [0, 4]`.\n\nStores 257 samples so interpolation can safely read `lut[idx + 1]` while indexing the 256 fractional segments across the interval.",
          "examples": [],
          "id": "arm_nn_tanh_lut_f16",
          "kind": "attribute",
          "language": "c",
          "members": [],
          "name": "arm_nn_tanh_lut_f16",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "const uint16_t arm_nn_tanh_lut_f16[257]",
          "source": {
            "line": 652,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L652"
          },
          "summary": "Quantized binary16 LUT for tanh(x) with x in [0, 4]."
        },
        {
          "description": "Reinterpret a 16-bit pattern as a float16.",
          "examples": [],
          "id": "arm_nn_softmax_fp16_from_bits",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_fp16_from_bits",
          "params": [
            {
              "description": "IEEE-754 binary16 bit pattern.",
              "direction": "in",
              "name": "bits",
              "type": "uint16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The float16 value with the bit pattern `bits`."
            }
          ],
          "signature": "static float16_t arm_nn_softmax_fp16_from_bits(uint16_t bits)",
          "source": {
            "line": 660,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L660"
          },
          "summary": "Reinterpret a 16-bit pattern as a float16."
        },
        {
          "description": "Floor of `x` as an int32_t.",
          "examples": [],
          "id": "arm_nn_softmax_floor_to_int_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_floor_to_int_f16",
          "params": [
            {
              "description": "Value to floor. Must be finite and within the int32_t range.",
              "direction": "in",
              "name": "x",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Largest int32_t not greater than `x`."
            }
          ],
          "signature": "static int32_t arm_nn_softmax_floor_to_int_f16(float16_t x)",
          "source": {
            "line": 677,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L677"
          },
          "summary": "Floor of x as an int32t."
        },
        {
          "description": "Compute `2^n` as a float16 by building the exponent field directly.",
          "examples": [],
          "id": "arm_nn_softmax_exp2i_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_exp2i_f16",
          "params": [
            {
              "description": "Integer exponent. Clamped to the normal float16 exponent range `[-14, 15]`.",
              "direction": "in",
              "name": "n",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`2^n` as a float16."
            }
          ],
          "signature": "static float16_t arm_nn_softmax_exp2i_f16(int32_t n)",
          "source": {
            "line": 690,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L690"
          },
          "summary": "Compute 2^n as a float16 by building the exponent field directly."
        },
        {
          "description": "Taylor/Estrin exp approximation for float16 softmax helpers.\n\nThe evaluation uses float32 intermediates to keep the approximation stable, but it is fully independent from the float32 softmax support tables.",
          "examples": [],
          "id": "arm_nn_softmax_exp_taylor_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_exp_taylor_f16",
          "params": [
            {
              "description": "Exponent argument. Clamped to `[-80, 80]` before evaluation.",
              "direction": "in",
              "name": "x",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Approximation of `exp(x)`."
            }
          ],
          "signature": "static float16_t arm_nn_softmax_exp_taylor_f16(float16_t x)",
          "source": {
            "line": 710,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L710"
          },
          "summary": "Taylor/Estrin exp approximation for float16 softmax helpers."
        },
        {
          "description": "LUT-based exp approximation for float16 softmax helpers.\n\nSplits `x * log2(e)` into an integer part handled by arm_nn_softmax_exp2i_f16() and a fractional part interpolated linearly from `arm_nn_exp2_lut_f16`, using float32 intermediates.",
          "examples": [],
          "id": "arm_nn_softmax_exp_lut_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_exp_lut_f16",
          "params": [
            {
              "description": "Exponent argument. Clamped to `[-80, 80]` before evaluation.",
              "direction": "in",
              "name": "x",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Approximation of `exp(x)`."
            }
          ],
          "signature": "static float16_t arm_nn_softmax_exp_lut_f16(float16_t x)",
          "source": {
            "line": 746,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L746"
          },
          "summary": "LUT-based exp approximation for float16 softmax helpers."
        },
        {
          "description": "Scalar exp approximation used by the float16 softmax paths.\n\nDispatches to arm_nn_softmax_exp_taylor_f16() when `ARM_NN_USE_EXP_TAYLOR` is defined and to arm_nn_softmax_exp_lut_f16() otherwise.",
          "examples": [],
          "id": "arm_nn_softmax_exp_scalar_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_exp_scalar_f16",
          "params": [
            {
              "description": "Exponent argument.",
              "direction": "in",
              "name": "x",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Approximation of `exp(x)`."
            }
          ],
          "signature": "static float16_t arm_nn_softmax_exp_scalar_f16(float16_t x)",
          "source": {
            "line": 787,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L787"
          },
          "summary": "Scalar exp approximation used by the float16 softmax paths."
        },
        {
          "description": "Copy a float16 vector.",
          "examples": [],
          "id": "arm_memcpy_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memcpy_f16",
          "params": [
            {
              "description": "Destination buffer.",
              "direction": "out",
              "name": "dst",
              "type": "float16_t *"
            },
            {
              "description": "Source buffer.",
              "direction": "in",
              "name": "src",
              "type": "const float16_t *"
            },
            {
              "description": "Number of elements to copy.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memcpy_f16(float16_t *dst, const float16_t *src, uint32_t block_size)",
          "source": {
            "line": 979,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L979"
          },
          "summary": "Copy a float16 vector."
        },
        {
          "description": "Set a float16 vector to a constant value.",
          "examples": [],
          "id": "arm_memset_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memset_f16",
          "params": [
            {
              "description": "Destination buffer.",
              "direction": "out",
              "name": "dst",
              "type": "float16_t *"
            },
            {
              "description": "Fill value.",
              "direction": "in",
              "name": "val",
              "type": "const float16_t"
            },
            {
              "description": "Number of elements to write.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memset_f16(float16_t *dst, const float16_t val, uint32_t block_size)",
          "source": {
            "line": 1002,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1002"
          },
          "summary": "Set a float16 vector to a constant value."
        },
        {
          "description": "Specialized NHWC depthwise `2x5` kernel (float16).",
          "examples": [],
          "id": "arm_nn_depthwise_conv2x5_nhwc_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv2x5_nhwc_f16",
          "params": [
            {
              "description": "Input tensor in NHWC layout with shape `[batches][2][in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of batches.",
              "direction": "in",
              "name": "batches",
              "type": "int32_t"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Channel multiplier; the output has `in_c * ch_mult` channels.",
              "direction": "in",
              "name": "ch_mult",
              "type": "int32_t"
            },
            {
              "description": "Depthwise weights with shape `[2][5][in_c * ch_mult]`.",
              "direction": "in",
              "name": "kernel",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `in_c * ch_mult` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output tensor in NHWC layout with shape `[batches][1][out_w][in_c * ch_mult]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Output width. Output position `ow` reads input columns `ow..ow+4`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            },
            {
              "description": "Lower clamp bound applied to `out`.",
              "direction": "in",
              "name": "act_min",
              "type": "float16_t"
            },
            {
              "description": "Upper clamp bound applied to `out`.",
              "direction": "in",
              "name": "act_max",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_depthwise_conv2x5_nhwc_f16(\n    const float16_t *x_nhwc,\n    int32_t batches,\n    int32_t in_c,\n    int32_t in_w,\n    int32_t ch_mult,\n    const float16_t *kernel,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_w,\n    float16_t act_min,\n    float16_t act_max\n)",
          "source": {
            "line": 1039,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1039"
          },
          "summary": "Specialized NHWC depthwise 2x5 kernel (float16)."
        },
        {
          "description": "Specialized NHWC depthwise 1D kernel for `k=3`, `ch_mult=1` (float32).",
          "examples": [],
          "id": "arm_nn_depthwise_conv1d_k3_nhwc_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv1d_k3_nhwc_f16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of input (and output) channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Depthwise weights with shape `[3][in_c]`.",
              "direction": "in",
              "name": "kernel",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `in_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][in_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_depthwise_conv1d_k3_nhwc_f16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float16_t *kernel,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_w\n)",
          "source": {
            "line": 1054,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1054"
          },
          "summary": "Specialized NHWC depthwise 1D kernel for k=3, chmult=1 (float32)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=5` (float32).\n\n:::note\nMVE leg: blockwise float16 accumulation (AmbiqAI/ns-cmsis-nn#586). Input channel c feeds lane c % 8 with 5 taps per channel step; above 32 taps per output, a lane's float16 partial covers at most 6 channel steps; each block's lanes are folded into float32 pair accumulators (arm_nn_f16_fold_pairs_f32), which are summed once (arm_nn_f16_pairs_sum_f32), the bias is added in float32 and the total rounds to float16 once. Up to 32 taps: float16 lanes, a float16 reduction and the bias added in float16, as before, in the order the compiler gives them (it may reorder them under -ffast-math); only the fold's order is fixed. The scalar leg accumulates in float32 (#449, #465).\n\n:::",
          "examples": [],
          "id": "arm_nn_conv1d_k5_nhwc_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k5_nhwc_f16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights with shape `[out_c][5][in_c]`.",
              "direction": "in",
              "name": "kernel",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+4`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k5_nhwc_f16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float16_t *kernel,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 1073,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1073"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=5 (float32)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=5` (float32).\n\n:::note\nMVE leg: blockwise float16 accumulation (AmbiqAI/ns-cmsis-nn#586). Input channel c feeds lane c % 8 with 5 taps per channel step; above 32 taps per output, a lane's float16 partial covers at most 6 channel steps; each block's lanes are folded into float32 pair accumulators (arm_nn_f16_fold_pairs_f32), which are summed once (arm_nn_f16_pairs_sum_f32), the bias is added in float32 and the total rounds to float16 once. Up to 32 taps: float16 lanes, a float16 reduction and the bias added in float16, as before, in the order the compiler gives them (it may reorder them under -ffast-math); only the fold's order is fixed. The scalar leg accumulates in float32 (#449, #465).\n\n:::\n\n:::note\nEvery MVE accumulator lane stays in float16 (no blockwise fold, AmbiqAI/ns-cmsis-nn#586); the scalar leg is the same as arm_nn_conv1d_k5_nhwc_f16.\n\n:::",
          "examples": [],
          "id": "arm_nn_conv1d_k5_nhwc_f16_acc16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k5_nhwc_f16_acc16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights with shape `[out_c][5][in_c]`.",
              "direction": "in",
              "name": "kernel",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+4`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k5_nhwc_f16_acc16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float16_t *kernel,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 1088,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1088"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=5 (float32)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=5` (float16, packed weights).\n\nThe packed kernel uses the same `NTxN` RHS layout as `arm_nn_mat_mult_nt_n_packed_f16`, i.e. `[(5 * in_c)][out_c_block_of_8]`.\n\n:::note\nAccumulation width per leg. Scalar leg (non-MVE builds and ARM_MATH_AUTOVECTORIZE): bias and products accumulate in float32 and round to float16 once at the store (AmbiqAI/ns-cmsis-nn#449, #465). MVE leg: blockwise (#586): a lane's float16 partial covers at most 6 input channels (30 taps, bias first) before it is widened into a float32 accumulator; one rounding at the store. Up to 32 taps (in_c <= 6) this is the float16-lane result.\n\n:::",
          "examples": [],
          "id": "arm_nn_conv1d_k5_packed_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k5_packed_f16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights packed in output-channel blocks of 8 as described above.",
              "direction": "in",
              "name": "kernel_packed",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+4`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k5_packed_f16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float16_t *kernel_packed,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 1118,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1118"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=5 (float16, packed weights)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=5` (float16, packed weights).\n\nThe packed kernel uses the same `NTxN` RHS layout as `arm_nn_mat_mult_nt_n_packed_f16`, i.e. `[(5 * in_c)][out_c_block_of_8]`.\n\n:::note\nAccumulation width per leg. Scalar leg (non-MVE builds and ARM_MATH_AUTOVECTORIZE): bias and products accumulate in float32 and round to float16 once at the store (AmbiqAI/ns-cmsis-nn#449, #465). MVE leg: blockwise (#586): a lane's float16 partial covers at most 6 input channels (30 taps, bias first) before it is widened into a float32 accumulator; one rounding at the store. Up to 32 taps (in_c <= 6) this is the float16-lane result.\n\n:::\n\n:::note\nEvery MVE accumulator lane stays in float16 (no blockwise fold, AmbiqAI/ns-cmsis-nn#586); the scalar leg is the same as arm_nn_conv1d_k5_packed_f16.\n\n:::",
          "examples": [],
          "id": "arm_nn_conv1d_k5_packed_f16_acc16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k5_packed_f16_acc16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights packed in output-channel blocks of 8 as described above.",
              "direction": "in",
              "name": "kernel_packed",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+4`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k5_packed_f16_acc16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float16_t *kernel_packed,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 1133,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1133"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=5 (float16, packed weights)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=3` (float32).\n\n:::note\nMVE leg: blockwise float16 accumulation (AmbiqAI/ns-cmsis-nn#586). Input channel c feeds lane c % 8 with 3 taps per channel step; above 32 taps per output, a lane's float16 partial covers at most 10 channel steps; each block's lanes are folded into float32 pair accumulators (arm_nn_f16_fold_pairs_f32), which are summed once (arm_nn_f16_pairs_sum_f32), the bias is added in float32 and the total rounds to float16 once. Up to 32 taps: float16 lanes, a float16 reduction and the bias added in float16, as before, in the order the compiler gives them (it may reorder them under -ffast-math); only the fold's order is fixed. The scalar leg accumulates in float32 (#449, #465).\n\n:::",
          "examples": [],
          "id": "arm_nn_conv1d_k3_nhwc_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k3_nhwc_f16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights with shape `[out_c][3][in_c]`.",
              "direction": "in",
              "name": "kernel",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k3_nhwc_f16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float16_t *kernel,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 1153,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1153"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=3 (float32)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=3` (float32).\n\n:::note\nMVE leg: blockwise float16 accumulation (AmbiqAI/ns-cmsis-nn#586). Input channel c feeds lane c % 8 with 3 taps per channel step; above 32 taps per output, a lane's float16 partial covers at most 10 channel steps; each block's lanes are folded into float32 pair accumulators (arm_nn_f16_fold_pairs_f32), which are summed once (arm_nn_f16_pairs_sum_f32), the bias is added in float32 and the total rounds to float16 once. Up to 32 taps: float16 lanes, a float16 reduction and the bias added in float16, as before, in the order the compiler gives them (it may reorder them under -ffast-math); only the fold's order is fixed. The scalar leg accumulates in float32 (#449, #465).\n\n:::\n\n:::note\nEvery MVE accumulator lane stays in float16 (no blockwise fold, AmbiqAI/ns-cmsis-nn#586); the scalar leg is the same as arm_nn_conv1d_k3_nhwc_f16.\n\n:::",
          "examples": [],
          "id": "arm_nn_conv1d_k3_nhwc_f16_acc16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k3_nhwc_f16_acc16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights with shape `[out_c][3][in_c]`.",
              "direction": "in",
              "name": "kernel",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k3_nhwc_f16_acc16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float16_t *kernel,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 1168,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1168"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=3 (float32)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=3` (float16, packed weights).\n\nThe packed kernel uses the same `NTxN` RHS layout as `arm_nn_mat_mult_nt_n_packed_f16`, i.e. `[(3 * in_c)][out_c_block_of_8]`.\n\n:::note\nAccumulation width per leg. Scalar leg (non-MVE builds and ARM_MATH_AUTOVECTORIZE): bias and products accumulate in float32 and round to float16 once at the store (AmbiqAI/ns-cmsis-nn#449, #465). MVE leg: blockwise (#586): a lane's float16 partial covers at most 10 input channels (30 taps, bias first) before it is widened into a float32 accumulator; one rounding at the store. Up to 32 taps (in_c <= 10) this is the float16-lane result.\n\n:::",
          "examples": [],
          "id": "arm_nn_conv1d_k3_packed_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k3_packed_f16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights packed in output-channel blocks of 8 as described above.",
              "direction": "in",
              "name": "kernel_packed",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k3_packed_f16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float16_t *kernel_packed,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 1198,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1198"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=3 (float16, packed weights)."
        },
        {
          "description": "Specialized NHWC 1D convolution kernel for `k=3` (float16, packed weights).\n\nThe packed kernel uses the same `NTxN` RHS layout as `arm_nn_mat_mult_nt_n_packed_f16`, i.e. `[(3 * in_c)][out_c_block_of_8]`.\n\n:::note\nAccumulation width per leg. Scalar leg (non-MVE builds and ARM_MATH_AUTOVECTORIZE): bias and products accumulate in float32 and round to float16 once at the store (AmbiqAI/ns-cmsis-nn#449, #465). MVE leg: blockwise (#586): a lane's float16 partial covers at most 10 input channels (30 taps, bias first) before it is widened into a float32 accumulator; one rounding at the store. Up to 32 taps (in_c <= 10) this is the float16-lane result.\n\n:::\n\n:::note\nEvery MVE accumulator lane stays in float16 (no blockwise fold, AmbiqAI/ns-cmsis-nn#586); the scalar leg is the same as arm_nn_conv1d_k3_packed_f16.\n\n:::",
          "examples": [],
          "id": "arm_nn_conv1d_k3_packed_f16_acc16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_conv1d_k3_packed_f16_acc16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Weights packed in output-channel blocks of 8 as described above.",
              "direction": "in",
              "name": "kernel_packed",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `out_c` elements. May be NULL.",
              "direction": "in",
              "name": "b",
              "type": "const float16_t *"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][out_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Number of output channels.",
              "direction": "in",
              "name": "out_c",
              "type": "int32_t"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `ow..ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_conv1d_k3_packed_f16_acc16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    const float16_t *kernel_packed,\n    const float16_t *b,\n    float16_t *out,\n    int32_t out_c,\n    int32_t out_w\n)",
          "source": {
            "line": 1213,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1213"
          },
          "summary": "Specialized NHWC 1D convolution kernel for k=3 (float16, packed weights)."
        },
        {
          "description": "Specialized NHWC max-pool 1D kernel for `k=3`, `s=3` (float16).",
          "examples": [],
          "id": "arm_nn_maxpool1d_k3s3_nhwc_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_maxpool1d_k3s3_nhwc_f16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][in_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `3*ow..3*ow+2`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_maxpool1d_k3s3_nhwc_f16(const float16_t *x_nhwc, int32_t in_c, int32_t in_w, float16_t *out, int32_t out_w)",
          "source": {
            "line": 1227,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1227"
          },
          "summary": "Specialized NHWC max-pool 1D kernel for k=3, s=3 (float16)."
        },
        {
          "description": "Specialized NHWC max-pool 1D kernel for `k=2`, `s=2` without output clamp (float16).",
          "examples": [],
          "id": "arm_nn_maxpool1d_k2s2_nhwc_noclip_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_maxpool1d_k2s2_nhwc_noclip_f16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][in_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `2*ow..2*ow+1`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_maxpool1d_k2s2_nhwc_noclip_f16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    float16_t *out,\n    int32_t out_w\n)",
          "source": {
            "line": 1238,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1238"
          },
          "summary": "Specialized NHWC max-pool 1D kernel for k=2, s=2 without output clamp (float16)."
        },
        {
          "description": "Specialized NHWC max-pool 1D kernel for `k=2`, `s=2` with clamp (float16).",
          "examples": [],
          "id": "arm_nn_maxpool1d_k2s2_nhwc_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_maxpool1d_k2s2_nhwc_f16",
          "params": [
            {
              "description": "Input row in NHWC layout with shape `[in_w][in_c]`.",
              "direction": "in",
              "name": "x_nhwc",
              "type": "const float16_t *"
            },
            {
              "description": "Number of channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Input width. Currently unused by the kernel.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Output row in NHWC layout with shape `[out_w][in_c]`.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            },
            {
              "description": "Output width. Output position `ow` reads input positions `2*ow..2*ow+1`.",
              "direction": "in",
              "name": "out_w",
              "type": "int32_t"
            },
            {
              "description": "Lower clamp bound applied to `out`.",
              "direction": "in",
              "name": "act_min",
              "type": "float16_t"
            },
            {
              "description": "Upper clamp bound applied to `out`.",
              "direction": "in",
              "name": "act_max",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_maxpool1d_k2s2_nhwc_f16(\n    const float16_t *x_nhwc,\n    int32_t in_c,\n    int32_t in_w,\n    float16_t *out,\n    int32_t out_w,\n    float16_t act_min,\n    float16_t act_max\n)",
          "source": {
            "line": 1249,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1249"
          },
          "summary": "Specialized NHWC max-pool 1D kernel for k=2, s=2 with clamp (float16)."
        },
        {
          "description": "Matrix multiply with non-transposed lhs and transposed rhs rows (float32).\n\n:::note\nAccumulation width per leg. MVE legs accumulate blockwise (AmbiqAI/ns-cmsis-nn#586, superseding the float16-lane choice of #417 / #446 for the MVE legs). Up to rhs_cols 32 nothing changes: per-k float16 lanes on the gather path (rhs_cols below the contiguous-K threshold), lane-partial sums then one float16 reduction plus the bias in float16 at rhs_cols 32. Above 32, on the contiguous-K path and the remainder rows, each lane (element k goes to lane k % 8) sums at most 32 of its own taps (256 elements) in float16; each block's lanes are then widened and lanes 2j and 2j+1 added in float32 into pair accumulator j (set by the first block, added to by later ones); the four pair accumulators are summed once as (0+1) + (2+3), so a single block sums ((0+1) + (2+3)) + ((4+5) + (6+7)), the bias is added in float32 and the total rounds to float16 once before the clamp. The float16 reduction up to rhs_cols 32 is ordered by the compiler, which may reorder it under -ffast-math; only the fold's order is fixed. arm_nn_mat_mult_nt_t_f16_acc16 keeps the float16 lanes and float16 reduction throughout. The scalar leg (non-MVE builds and ARM_MATH_AUTOVECTORIZE) accumulates bias and every product in float32 and rounds to float16 once before the clamp (AmbiqAI/ns-cmsis-nn#449, #457).\n\n:::",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_t_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_t_f16",
          "params": [
            {
              "description": "Left-hand matrix stored row-major.",
              "direction": "in",
              "name": "lhs",
              "type": "const float16_t *"
            },
            {
              "description": "Right-hand matrix stored row-major, one row per output channel.",
              "direction": "in",
              "name": "rhs",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector.",
              "direction": "in",
              "name": "bias",
              "type": "const float16_t *"
            },
            {
              "description": "Output matrix.",
              "direction": "out",
              "name": "dst",
              "type": "float16_t *"
            },
            {
              "description": "Number of rows in `lhs`.",
              "direction": "in",
              "name": "lhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Number of rows in `rhs`.",
              "direction": "in",
              "name": "rhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Number of columns in `rhs`.",
              "direction": "in",
              "name": "rhs_cols",
              "type": "int32_t"
            },
            {
              "description": "Output row stride, expressed in elements.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "int32_t"
            },
            {
              "description": "Lower clamp bound.",
              "direction": "in",
              "name": "activation_min",
              "type": "float16_t"
            },
            {
              "description": "Upper clamp bound.",
              "direction": "in",
              "name": "activation_max",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`ARM_CMSIS_NN_SUCCESS` on success or `ARM_CMSIS_NN_ARG_ERROR` on invalid arguments."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_t_f16(\n    const float16_t *lhs,\n    const float16_t *rhs,\n    const float16_t *bias,\n    float16_t *dst,\n    int32_t lhs_rows,\n    int32_t rhs_rows,\n    int32_t rhs_cols,\n    int32_t row_address_offset,\n    float16_t activation_min,\n    float16_t activation_max\n)",
          "source": {
            "line": 1273,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1273"
          },
          "summary": "Matrix multiply with non-transposed lhs and transposed rhs rows (float32)."
        },
        {
          "description": "arm_nn_mat_mult_nt_t_f16 with every MVE accumulator lane in float16 (no blockwise fold).\n\nSame arguments, return codes and scalar leg as arm_nn_mat_mult_nt_t_f16; see its accumulation note.",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_t_f16_acc16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_t_f16_acc16",
          "params": [
            {
              "description": "Left-hand matrix, row-major `[lhs_rows, rhs_cols]`.",
              "direction": "in",
              "name": "lhs",
              "type": "const float16_t *"
            },
            {
              "description": "Right-hand matrix, row-major `[rhs_rows, rhs_cols]` (transposed operand).",
              "direction": "in",
              "name": "rhs",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector of `rhs_rows` elements.",
              "direction": "in",
              "name": "bias",
              "type": "const float16_t *"
            },
            {
              "description": "Output matrix.",
              "direction": "out",
              "name": "dst",
              "type": "float16_t *"
            },
            {
              "description": "Number of rows in `lhs`.",
              "direction": "in",
              "name": "lhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Number of rows in `rhs`.",
              "direction": "in",
              "name": "rhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Shared reduction dimension `K`.",
              "direction": "in",
              "name": "rhs_cols",
              "type": "int32_t"
            },
            {
              "description": "Output row stride, expressed in elements.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "int32_t"
            },
            {
              "description": "Lower clamp bound.",
              "direction": "in",
              "name": "activation_min",
              "type": "float16_t"
            },
            {
              "description": "Upper clamp bound.",
              "direction": "in",
              "name": "activation_max",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`ARM_CMSIS_NN_SUCCESS` on success or `ARM_CMSIS_NN_ARG_ERROR` on invalid arguments."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_t_f16_acc16(\n    const float16_t *lhs,\n    const float16_t *rhs,\n    const float16_t *bias,\n    float16_t *dst,\n    int32_t lhs_rows,\n    int32_t rhs_rows,\n    int32_t rhs_cols,\n    int32_t row_address_offset,\n    float16_t activation_min,\n    float16_t activation_max\n)",
          "source": {
            "line": 1301,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1301"
          },
          "summary": "armnnmatmultnttf16 with every MVE accumulator lane in float16 (no blockwise fold)."
        },
        {
          "description": "Matrix multiply with non-transposed lhs and packed non-transposed rhs (float16).\n\n:::note\nOn non-MVE builds the output clamp is the bit-classified scalar clamp of #380, so a NaN accumulator (a NaN in `lhs`, `rhs_packed` or `bias`) propagates to `dst` at every optimization level on the gated toolchains, including the shipped -Ofast. On MVE builds the clamp is vmaxnmq/vminnmq with no NaN restore, so a NaN resolves to a clamp bound there instead.\n\n:::\n\n:::note\nAccumulation width per leg: the MVE leg accumulates blockwise (AmbiqAI/ns-cmsis-nn#586): one lane per output column, per-k, the bias opening the first block; every 32 k the float16 partial is widened exactly into per-lane float32 accumulators, which round to float16 once before the clamp (rhs_cols up to 32: exactly the float16-lane result; arm_nn_mat_mult_nt_n_packed_f16_acc16 keeps float16 lanes throughout). The scalar leg (non-MVE builds and ARM_MATH_AUTOVECTORIZE) accumulates bias and every product in float32 and rounds to float16 once before the clamp (AmbiqAI/ns-cmsis-nn#449, #457).\n\n:::",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_n_packed_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_n_packed_f16",
          "params": [
            {
              "description": "Left-hand matrix stored row-major with logical shape `[lhs_rows, rhs_cols]`.",
              "direction": "in",
              "name": "lhs",
              "type": "const float16_t *"
            },
            {
              "description": "Right-hand matrix with logical shape `[rhs_cols, rhs_rows]`, packed in column blocks of 8. The final block uses the same packed stride and inactive tail lanes are ignored.",
              "direction": "in",
              "name": "rhs_packed",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector.",
              "direction": "in",
              "name": "bias",
              "type": "const float16_t *"
            },
            {
              "description": "Output matrix.",
              "direction": "out",
              "name": "dst",
              "type": "float16_t *"
            },
            {
              "description": "Number of rows in `lhs`.",
              "direction": "in",
              "name": "lhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Number of logical output columns in the unpacked rhs matrix.",
              "direction": "in",
              "name": "rhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Shared reduction dimension `K`.",
              "direction": "in",
              "name": "rhs_cols",
              "type": "int32_t"
            },
            {
              "description": "Output row stride, expressed in elements.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "int32_t"
            },
            {
              "description": "Lower clamp bound.",
              "direction": "in",
              "name": "activation_min",
              "type": "float16_t"
            },
            {
              "description": "Upper clamp bound.",
              "direction": "in",
              "name": "activation_max",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`ARM_CMSIS_NN_SUCCESS` on success or `ARM_CMSIS_NN_ARG_ERROR` on invalid arguments."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_n_packed_f16(\n    const float16_t *lhs,\n    const float16_t *rhs_packed,\n    const float16_t *bias,\n    float16_t *dst,\n    int32_t lhs_rows,\n    int32_t rhs_rows,\n    int32_t rhs_cols,\n    int32_t row_address_offset,\n    float16_t activation_min,\n    float16_t activation_max\n)",
          "source": {
            "line": 1342,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1342"
          },
          "summary": "Matrix multiply with non-transposed lhs and packed non-transposed rhs (float16)."
        },
        {
          "description": "arm_nn_mat_mult_nt_n_packed_f16 with every MVE accumulator lane in float16 (no blockwise fold).\n\nSame arguments, return codes and scalar leg as arm_nn_mat_mult_nt_n_packed_f16; see its accumulation note.",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_n_packed_f16_acc16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_n_packed_f16_acc16",
          "params": [
            {
              "description": "Left-hand matrix stored row-major with logical shape `[lhs_rows, rhs_cols]`.",
              "direction": "in",
              "name": "lhs",
              "type": "const float16_t *"
            },
            {
              "description": "Right-hand matrix packed in column blocks of 8.",
              "direction": "in",
              "name": "rhs_packed",
              "type": "const float16_t *"
            },
            {
              "description": "Optional bias vector.",
              "direction": "in",
              "name": "bias",
              "type": "const float16_t *"
            },
            {
              "description": "Output matrix.",
              "direction": "out",
              "name": "dst",
              "type": "float16_t *"
            },
            {
              "description": "Number of rows in `lhs`.",
              "direction": "in",
              "name": "lhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Number of logical output columns in the unpacked rhs matrix.",
              "direction": "in",
              "name": "rhs_rows",
              "type": "int32_t"
            },
            {
              "description": "Shared reduction dimension `K`.",
              "direction": "in",
              "name": "rhs_cols",
              "type": "int32_t"
            },
            {
              "description": "Output row stride, expressed in elements.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "int32_t"
            },
            {
              "description": "Lower clamp bound.",
              "direction": "in",
              "name": "activation_min",
              "type": "float16_t"
            },
            {
              "description": "Upper clamp bound.",
              "direction": "in",
              "name": "activation_max",
              "type": "float16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`ARM_CMSIS_NN_SUCCESS` on success or `ARM_CMSIS_NN_ARG_ERROR` on invalid arguments."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_n_packed_f16_acc16(\n    const float16_t *lhs,\n    const float16_t *rhs_packed,\n    const float16_t *bias,\n    float16_t *dst,\n    int32_t lhs_rows,\n    int32_t rhs_rows,\n    int32_t rhs_cols,\n    int32_t row_address_offset,\n    float16_t activation_min,\n    float16_t activation_max\n)",
          "source": {
            "line": 1370,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1370"
          },
          "summary": "armnnmatmultntnpackedf16 with every MVE accumulator lane in float16 (no blockwise fold)."
        },
        {
          "description": "Update LSTM function for an iteration step using float16 input, output and state.",
          "examples": [],
          "id": "arm_nn_lstm_step_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_lstm_step_f16",
          "params": [
            {
              "description": "Data input pointer.",
              "direction": "in",
              "name": "data_in",
              "type": "const float16_t *"
            },
            {
              "description": "Hidden state / recurrent input pointer. May be NULL for the first step.",
              "direction": "in",
              "name": "hidden_in",
              "type": "const float16_t *"
            },
            {
              "description": "Hidden state / recurrent output pointer.",
              "direction": "out",
              "name": "hidden_out",
              "type": "float16_t *"
            },
            {
              "description": "Struct containing all information about the LSTM operator.",
              "direction": "in",
              "name": "params",
              "type": "const cmsis_nn_lstm_params_f16 *"
            },
            {
              "description": "Struct containing pointers to mutable cell-state storage.",
              "direction": "inout",
              "name": "buffers",
              "type": "cmsis_nn_lstm_context_f16 *"
            },
            {
              "description": "Number of timesteps between consecutive batches.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "ARM_CMSIS_NN_SUCCESS on success, or ARM_CMSIS_NN_ARG_ERROR on invalid arguments (NULL data_in/hidden_out/params/buffers or buffers->cell_state, batch_offset <= 0)."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_lstm_step_f16(\n    const float16_t *data_in,\n    const float16_t *hidden_in,\n    float16_t *hidden_out,\n    const cmsis_nn_lstm_params_f16 *params,\n    cmsis_nn_lstm_context_f16 *buffers,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 1394,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1394"
          },
          "summary": "Update LSTM function for an iteration step using float16 input, output and state."
        },
        {
          "description": "Update GRU function for a single iteration step using float16 data.",
          "examples": [],
          "id": "arm_nn_gru_step_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_gru_step_f16",
          "params": [
            {
              "description": "Data input pointer for this time step.",
              "direction": "in",
              "name": "data_in",
              "type": "const float16_t *"
            },
            {
              "description": "Recurrent input pointer. NULL for the first step (h_prev = 0).",
              "direction": "in",
              "name": "hidden_in",
              "type": "const float16_t *"
            },
            {
              "description": "Hidden-state output pointer for this time step.",
              "direction": "out",
              "name": "hidden_out",
              "type": "float16_t *"
            },
            {
              "description": "Struct describing the GRU operator.",
              "direction": "in",
              "name": "params",
              "type": "const cmsis_nn_gru_params_f16 *"
            },
            {
              "description": "Scratch buffers. temp1 (>= hidden_size) is required when reset_after == 0.",
              "direction": "inout",
              "name": "buffers",
              "type": "cmsis_nn_gru_context_f16 *"
            },
            {
              "description": "Number of timesteps between consecutive batches.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "ARM_CMSIS_NN_SUCCESS on success, or ARM_CMSIS_NN_ARG_ERROR on invalid arguments (NULL data_in/hidden_out/params, batch_offset <= 0, or missing temp1 when reset_after == 0)."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_gru_step_f16(\n    const float16_t *data_in,\n    const float16_t *hidden_in,\n    float16_t *hidden_out,\n    const cmsis_nn_gru_params_f16 *params,\n    cmsis_nn_gru_context_f16 *buffers,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 1414,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1414"
          },
          "summary": "Update GRU function for a single iteration step using float16 data."
        },
        {
          "description": "Pack a single convolution patch into one row of a contiguous float32 patch matrix.\n\nDevelopers familiar with im2row/im2col terminology can think of this as packing one output patch into one row.",
          "examples": [],
          "id": "arm_nn_pack_conv_patch_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_pack_conv_patch_f16",
          "params": [
            {
              "description": "Input tensor for one batch in NHWC layout with shape `[in_h][in_w][in_c]`.",
              "direction": "in",
              "name": "input",
              "type": "const float16_t *"
            },
            {
              "description": "Input height.",
              "direction": "in",
              "name": "in_h",
              "type": "int32_t"
            },
            {
              "description": "Input width.",
              "direction": "in",
              "name": "in_w",
              "type": "int32_t"
            },
            {
              "description": "Number of input channels.",
              "direction": "in",
              "name": "in_c",
              "type": "int32_t"
            },
            {
              "description": "Kernel height.",
              "direction": "in",
              "name": "kernel_h",
              "type": "int32_t"
            },
            {
              "description": "Kernel width.",
              "direction": "in",
              "name": "kernel_w",
              "type": "int32_t"
            },
            {
              "description": "Vertical stride.",
              "direction": "in",
              "name": "stride_h",
              "type": "int32_t"
            },
            {
              "description": "Horizontal stride.",
              "direction": "in",
              "name": "stride_w",
              "type": "int32_t"
            },
            {
              "description": "Top padding.",
              "direction": "in",
              "name": "pad_h",
              "type": "int32_t"
            },
            {
              "description": "Left padding.",
              "direction": "in",
              "name": "pad_w",
              "type": "int32_t"
            },
            {
              "description": "Vertical dilation.",
              "direction": "in",
              "name": "dilation_h",
              "type": "int32_t"
            },
            {
              "description": "Horizontal dilation.",
              "direction": "in",
              "name": "dilation_w",
              "type": "int32_t"
            },
            {
              "description": "Output row index of the patch to pack.",
              "direction": "in",
              "name": "out_y",
              "type": "int32_t"
            },
            {
              "description": "Output column index of the patch to pack.",
              "direction": "in",
              "name": "out_x",
              "type": "int32_t"
            },
            {
              "description": "Value written for taps that fall outside the input.",
              "direction": "in",
              "name": "pad_value",
              "type": "float16_t"
            },
            {
              "description": "Destination row of `kernel_h * kernel_w * in_c` elements, ordered `[kernel_h][kernel_w][in_c]`.",
              "direction": "out",
              "name": "patch_row",
              "type": "float16_t *"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_pack_conv_patch_f16(\n    const float16_t *input,\n    int32_t in_h,\n    int32_t in_w,\n    int32_t in_c,\n    int32_t kernel_h,\n    int32_t kernel_w,\n    int32_t stride_h,\n    int32_t stride_w,\n    int32_t pad_h,\n    int32_t pad_w,\n    int32_t dilation_h,\n    int32_t dilation_w,\n    int32_t out_y,\n    int32_t out_x,\n    float16_t pad_value,\n    float16_t *patch_row\n)",
          "source": {
            "line": 1424,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1424"
          },
          "summary": "Pack a single convolution patch into one row of a contiguous float32 patch matrix."
        },
        {
          "description": "Specialized softmax helper for a single float16 row of length 2.",
          "examples": [],
          "id": "arm_nn_softmax_1x2_f16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_1x2_f16",
          "params": [
            {
              "description": "Pointer to two contiguous float16 input values.",
              "direction": "in",
              "name": "in",
              "type": "const float16_t *"
            },
            {
              "description": "Pointer to two contiguous float16 output values.",
              "direction": "out",
              "name": "out",
              "type": "float16_t *"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_softmax_1x2_f16(const float16_t *in, float16_t *out)",
          "source": {
            "line": 1447,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1447"
          },
          "summary": "Specialized softmax helper for a single float16 row of length 2."
        },
        {
          "description": "Update LSTM function for an iteration step using float32 input, output and state.",
          "examples": [],
          "id": "arm_nn_lstm_step_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_lstm_step_f32",
          "params": [
            {
              "description": "Data input pointer.",
              "direction": "in",
              "name": "data_in",
              "type": "const float32_t *"
            },
            {
              "description": "Hidden state / recurrent input pointer. May be NULL for the first step.",
              "direction": "in",
              "name": "hidden_in",
              "type": "const float32_t *"
            },
            {
              "description": "Hidden state / recurrent output pointer.",
              "direction": "out",
              "name": "hidden_out",
              "type": "float32_t *"
            },
            {
              "description": "Struct containing all information about the LSTM operator.",
              "direction": "in",
              "name": "params",
              "type": "const cmsis_nn_lstm_params_f32 *"
            },
            {
              "description": "Struct containing pointers to mutable cell-state storage.",
              "direction": "inout",
              "name": "buffers",
              "type": "cmsis_nn_lstm_context_f32 *"
            },
            {
              "description": "Number of timesteps between consecutive batches.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "ARM_CMSIS_NN_SUCCESS on success, or ARM_CMSIS_NN_ARG_ERROR on invalid arguments (NULL data_in/hidden_out/params/buffers or buffers->cell_state, batch_offset <= 0)."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_lstm_step_f32(\n    const float32_t *data_in,\n    const float32_t *hidden_in,\n    float32_t *hidden_out,\n    const cmsis_nn_lstm_params_f32 *params,\n    cmsis_nn_lstm_context_f32 *buffers,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 1466,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1466"
          },
          "summary": "Update LSTM function for an iteration step using float32 input, output and state."
        },
        {
          "description": "Update GRU function for a single iteration step using float32 data.",
          "examples": [],
          "id": "arm_nn_gru_step_f32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_gru_step_f32",
          "params": [
            {
              "description": "Data input pointer for this time step.",
              "direction": "in",
              "name": "data_in",
              "type": "const float32_t *"
            },
            {
              "description": "Recurrent input pointer. NULL for the first step (h_prev = 0).",
              "direction": "in",
              "name": "hidden_in",
              "type": "const float32_t *"
            },
            {
              "description": "Hidden-state output pointer for this time step.",
              "direction": "out",
              "name": "hidden_out",
              "type": "float32_t *"
            },
            {
              "description": "Struct describing the GRU operator.",
              "direction": "in",
              "name": "params",
              "type": "const cmsis_nn_gru_params_f32 *"
            },
            {
              "description": "Scratch buffers. temp1 (>= hidden_size) is required when reset_after == 0.",
              "direction": "inout",
              "name": "buffers",
              "type": "cmsis_nn_gru_context_f32 *"
            },
            {
              "description": "Number of timesteps between consecutive batches.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "ARM_CMSIS_NN_SUCCESS on success, or ARM_CMSIS_NN_ARG_ERROR on invalid arguments (NULL data_in/hidden_out/params, batch_offset <= 0, or missing temp1 when reset_after == 0)."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_gru_step_f32(\n    const float32_t *data_in,\n    const float32_t *hidden_in,\n    float32_t *hidden_out,\n    const cmsis_nn_gru_params_f32 *params,\n    cmsis_nn_gru_context_f32 *buffers,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 1486,
            "path": "Include/arm_nnsupportfunctions_flt.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions_flt.h#L1486"
          },
          "summary": "Update GRU function for a single iteration step using float32 data."
        }
      ]
    }
  ],
  "name": "heliaCORE"
}
