{
  "$schema": "https://ambiqai.github.io/helia-ui/schema/reference-model-1.json",
  "generatedFrom": {
    "sourceCommit": "5f3fed9f21a57390cc7f00f77a37db8f5f110cb8",
    "tool": "doxyref",
    "version": "1.17.0"
  },
  "language": "c",
  "modules": [
    {
      "description": "",
      "name": "arm_nnsupportfunctions.h",
      "path": "heliaCORE.arm_nnsupportfunctions",
      "submodules": [],
      "summary": "",
      "symbols": [
        {
          "description": "",
          "examples": [],
          "id": "USE_FAST_DW_CONV_S16_FUNCTION",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "USE_FAST_DW_CONV_S16_FUNCTION",
          "params": [
            {
              "description": "",
              "name": "dw_conv_params"
            },
            {
              "description": "",
              "name": "filter_dims"
            },
            {
              "description": "",
              "name": "input_dims"
            },
            {
              "description": "",
              "name": "output_dims"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define USE_FAST_DW_CONV_S16_FUNCTION(dw_conv_params, filter_dims, input_dims, output_dims) (dw_conv_params->ch_mult == 1 && \\ arm_nn_dw_conv_opt_dilation_supported(dw_conv_params, input_dims, filter_dims, output_dims) && \\ filter_dims->w * filter_dims->h < 512)",
          "source": {
            "line": 45,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L45"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "LEFT_SHIFT",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "LEFT_SHIFT",
          "params": [
            {
              "description": "",
              "name": "_shift"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define LEFT_SHIFT(_shift) (_shift > 0 ? _shift : 0)",
          "source": {
            "line": 50,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L50"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "RIGHT_SHIFT",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "RIGHT_SHIFT",
          "params": [
            {
              "description": "",
              "name": "_shift"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define RIGHT_SHIFT(_shift) (_shift > 0 ? 0 : -_shift)",
          "source": {
            "line": 51,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L51"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "MASK_IF_ZERO",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "MASK_IF_ZERO",
          "params": [
            {
              "description": "",
              "name": "x"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define MASK_IF_ZERO(x) (x) == 0 ? ~0 : 0",
          "source": {
            "line": 52,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L52"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "MASK_IF_NON_ZERO",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "MASK_IF_NON_ZERO",
          "params": [
            {
              "description": "",
              "name": "x"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define MASK_IF_NON_ZERO(x) (x) != 0 ? ~0 : 0",
          "source": {
            "line": 53,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L53"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "SELECT_USING_MASK",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "SELECT_USING_MASK",
          "params": [
            {
              "description": "",
              "name": "mask"
            },
            {
              "description": "",
              "name": "a"
            },
            {
              "description": "",
              "name": "b"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define SELECT_USING_MASK(mask, a, b) ((mask) & (a)) ^ (~(mask) & (b))",
          "source": {
            "line": 54,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L54"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "ARM_NN_MAX",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_MAX",
          "params": [
            {
              "description": "",
              "name": "A"
            },
            {
              "description": "",
              "name": "B"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_MAX(A, B) ((A) > (B) ? (A) : (B))",
          "source": {
            "line": 57,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L57"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "ARM_NN_MIN",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_MIN",
          "params": [
            {
              "description": "",
              "name": "A"
            },
            {
              "description": "",
              "name": "B"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_MIN(A, B) ((A) < (B) ? (A) : (B))",
          "source": {
            "line": 58,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L58"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "ARM_NN_CLAMP",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_CLAMP",
          "params": [
            {
              "description": "",
              "name": "x"
            },
            {
              "description": "",
              "name": "h"
            },
            {
              "description": "",
              "name": "l"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_CLAMP(x, h, l) ARM_NN_MAX(ARM_NN_MIN((x), (h)), (l))",
          "source": {
            "line": 59,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L59"
          },
          "summary": ""
        },
        {
          "description": "Minimum of two scalar f16 values.\n\nWith ARM_NN_F16_CMOV_WORKAROUND this is IEEE 754 minNum via VMINNM.F16: a NaN operand is suppressed and the non-NaN operand wins. The scalar fallback is ARM_NN_MIN, an ordered compare, so its NaN handling depends on operand order: a NaN `b` is returned, a NaN `a` is not. Do not rely on NaN suppression on non-MVE builds.",
          "examples": [],
          "id": "arm_nn_min_f16h",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_min_f16h",
          "params": [
            {
              "description": "First operand",
              "direction": "in",
              "name": "a",
              "type": "_Float16"
            },
            {
              "description": "Second operand",
              "direction": "in",
              "name": "b",
              "type": "_Float16"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The smaller of `a` and `b`"
            }
          ],
          "signature": "static _Float16 arm_nn_min_f16h(_Float16 a, _Float16 b)",
          "source": {
            "line": 98,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L98"
          },
          "summary": "Minimum of two scalar f16 values."
        },
        {
          "description": "Maximum of two scalar f16 values.\n\nWith ARM_NN_F16_CMOV_WORKAROUND this is IEEE 754 maxNum via VMAXNM.F16: a NaN operand is suppressed and the non-NaN operand wins. The scalar fallback is ARM_NN_MAX, an ordered compare, so its NaN handling depends on operand order: a NaN `b` is returned, a NaN `a` is not. Do not rely on NaN suppression on non-MVE builds.",
          "examples": [],
          "id": "arm_nn_max_f16h",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_max_f16h",
          "params": [
            {
              "description": "First operand",
              "direction": "in",
              "name": "a",
              "type": "_Float16"
            },
            {
              "description": "Second operand",
              "direction": "in",
              "name": "b",
              "type": "_Float16"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The larger of `a` and `b`"
            }
          ],
          "signature": "static _Float16 arm_nn_max_f16h(_Float16 a, _Float16 b)",
          "source": {
            "line": 121,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L121"
          },
          "summary": "Maximum of two scalar f16 values."
        },
        {
          "description": "Returns `x` when `x` is NaN, otherwise `y`.\n\nBoth the NaN test and the select are performed on the bit patterns: the test is (bits & 0x7FFF) > 0x7C00 (all-ones exponent, non-zero mantissa), which is integer arithmetic that -ffinite-math-only (implied by the shipped -Ofast) has no license to fold, unlike the former floating-point self-compare `x != x` (#333 / #334); the bit-pattern select neither expands to an HFmode conditional move (PR target/118460) nor quiets/retags the NaN payload. This helper backs the f16 elementwise clamp and, via arm_nn_clamp_scalar_f16 / arm_nn_clamp_propagate_nan_f16h, the other f16 scalar clamp users  arm_svdf_f16, arm_max_pool_f16 / arm_avg_pool_f16, the packed f16 matmul (arm_nn_mat_mult_nt_n_packed_f16), the scalar f16 RELU/RELU6/LEAKY_RELU activation legs, and arm_nn_vector_clamp_f16's scalar leg (conv/depthwise/transpose-conv f16, the 3x3 depthwise, and arm_nn_maxpool1d_f16)  so wherever that scalar clamp runs, a NaN passes through it at every optimization level on the gated toolchains. That is a guarantee about the clamp, not the whole kernel: which builds run the scalar clamp, and whether a NaN survives the rest of the kernel to reach it, is per kernel  several of these callers clamp with vmaxnmq/vminnmq on MVE builds (a NaN resolves to a bound there), and arm_max_pool_f16's max reduction drops a NaN before the clamp. The kernels with a NaN\n\n:::note\n(svdf, max/avg pool, packed matmul, activation) state their exact scope there; the arm_nn_vector_clamp_f16 family is covered by a test assertion in the transpose-conv f16 suite rather than per-kernel notes. The cortex-m55 MVE RELU/RELU6 f16 legs reach the same guarantee by the vector form of this idiom rather than by calling this helper: they restore NaN lanes with arm_nn_max_propagate_nan_mve_f16 / arm_nn_clamp_propagate_nan_mve_f16 (#382). The same idiom (bit-classified select) appears in arm_prelu_f16, which does not call this helper.\n\n:::",
          "examples": [],
          "id": "arm_nn_propagate_nan_f16h",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_propagate_nan_f16h",
          "params": [
            {
              "description": "Value whose NaN-ness selects the result. Returned unchanged when it is NaN.",
              "direction": "in",
              "name": "x",
              "type": "_Float16"
            },
            {
              "description": "Value returned when `x` is not NaN",
              "direction": "in",
              "name": "y",
              "type": "_Float16"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`x` if `x` is NaN, otherwise `y`"
            }
          ],
          "signature": "static _Float16 arm_nn_propagate_nan_f16h(_Float16 x, _Float16 y)",
          "source": {
            "line": 167,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L167"
          },
          "summary": "Returns x when x is NaN, otherwise y."
        },
        {
          "description": "Drop-in equivalent of `ARM_NN_CLAMP(x, h, l)` for scalar _Float16 operands.\n\nIncludes the macro's NaN behaviour: `ARM_NN_MIN(NaN, h)` is h, so a NaN input resolves to the high bound, exactly as the macro does. Use arm_nn_clamp_propagate_nan_f16h() where TFLite NaN propagation is required.",
          "examples": [],
          "id": "arm_nn_clamp_f16h",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_clamp_f16h",
          "params": [
            {
              "description": "Value to clamp",
              "direction": "in",
              "name": "x",
              "type": "_Float16"
            },
            {
              "description": "Upper bound",
              "direction": "in",
              "name": "h",
              "type": "_Float16"
            },
            {
              "description": "Lower bound",
              "direction": "in",
              "name": "l",
              "type": "_Float16"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`x` clamped to [`l`, `h`]"
            }
          ],
          "signature": "static _Float16 arm_nn_clamp_f16h(_Float16 x, _Float16 h, _Float16 l)",
          "source": {
            "line": 193,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L193"
          },
          "summary": "Drop-in equivalent of ARMNNCLAMP(x, h, l) for scalar Float16 operands."
        },
        {
          "description": "Scalar f16 clamp with TFLite NaN semantics: NaN passes through unchanged.\n\nMirrors the MVE idiom in arm_nn_clamp_propagate_nan_mve_f16() (lower bound first, then upper bound, then restore NaN lanes). The NaN restore in arm_nn_propagate_nan_f16h() tests the integer bit pattern, so it holds at every optimization level including the shipped -Ofast; see #333 / #334. Bounds are assumed ordered (l <= h); inverted bounds are unspecified.",
          "examples": [],
          "id": "arm_nn_clamp_propagate_nan_f16h",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_clamp_propagate_nan_f16h",
          "params": [
            {
              "description": "Value to clamp",
              "direction": "in",
              "name": "x",
              "type": "_Float16"
            },
            {
              "description": "Lower bound",
              "direction": "in",
              "name": "l",
              "type": "_Float16"
            },
            {
              "description": "Upper bound",
              "direction": "in",
              "name": "h",
              "type": "_Float16"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`x` clamped to [`l`, `h`], or `x` itself when it is NaN"
            }
          ],
          "signature": "static _Float16 arm_nn_clamp_propagate_nan_f16h(_Float16 x, _Float16 l, _Float16 h)",
          "source": {
            "line": 212,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L212"
          },
          "summary": "Scalar f16 clamp with TFLite NaN semantics: NaN passes through unchanged."
        },
        {
          "description": "Absolute value of a scalar f16 value.",
          "examples": [],
          "id": "arm_nn_abs_f16h",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_abs_f16h",
          "params": [
            {
              "description": "Input value",
              "direction": "in",
              "name": "x",
              "type": "_Float16"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "|`x`|"
            }
          ],
          "signature": "static _Float16 arm_nn_abs_f16h(_Float16 x)",
          "source": {
            "line": 224,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L224"
          },
          "summary": "Absolute value of a scalar f16 value."
        },
        {
          "description": "",
          "examples": [],
          "id": "ARM_NN_ROUND_UP",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_ROUND_UP",
          "params": [
            {
              "description": "",
              "name": "x"
            },
            {
              "description": "",
              "name": "multiple"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_ROUND_UP(x, multiple) ((((x) + (multiple) - 1) / (multiple)) * (multiple))",
          "source": {
            "line": 236,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L236"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "REDUCE_MULTIPLIER",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "REDUCE_MULTIPLIER",
          "params": [
            {
              "description": "",
              "name": "_mult"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define REDUCE_MULTIPLIER(_mult) ((_mult < 0x7FFF0000) ? ((_mult + (1 << 15)) >> 16) : 0x7FFF)",
          "source": {
            "line": 237,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L237"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "CH_IN_BLOCK_MVE",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "CH_IN_BLOCK_MVE",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define CH_IN_BLOCK_MVE (124)",
          "source": {
            "line": 245,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L245"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "S4_CH_IN_BLOCK_MVE",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "S4_CH_IN_BLOCK_MVE",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define S4_CH_IN_BLOCK_MVE (124)",
          "source": {
            "line": 250,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L250"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "MAX_COL_COUNT",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "MAX_COL_COUNT",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define MAX_COL_COUNT (512)",
          "source": {
            "line": 254,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L254"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "REVERSE_TCOL_EFFICIENT_THRESHOLD",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "REVERSE_TCOL_EFFICIENT_THRESHOLD",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define REVERSE_TCOL_EFFICIENT_THRESHOLD (16)",
          "source": {
            "line": 258,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L258"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "CONVERT_DW_CONV_WITH_ONE_INPUT_CH_AND_OUTPUT_CH_ABOVE_THRESHOLD",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "CONVERT_DW_CONV_WITH_ONE_INPUT_CH_AND_OUTPUT_CH_ABOVE_THRESHOLD",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define CONVERT_DW_CONV_WITH_ONE_INPUT_CH_AND_OUTPUT_CH_ABOVE_THRESHOLD (1)",
          "source": {
            "line": 266,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L266"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "OPTIONAL_RESTRICT_KEYWORD",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "OPTIONAL_RESTRICT_KEYWORD",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define OPTIONAL_RESTRICT_KEYWORD",
          "source": {
            "line": 272,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L272"
          },
          "summary": ""
        },
        {
          "description": "Fold one dimension into a running buffer-size product, reporting overflow as -1.\n\nBuffer-size queries return an int32_t byte count, so the product of the dimensions they multiply has to be rejected as soon as it cannot fit. Folding one factor at a time keeps the accumulator bounded: an accumulator already known to be <= INT32_MAX times a factor <= INT32_MAX cannot exceed about 2^62, so the int64_t accumulator itself never wraps. Chaining raw (int64_t) casts across three or more int32_t dims does not have that property - 65536 * 65536 * 65536 * 65536 is exactly 2^64 and folds back to 0, which would sail through a trailing \"> INT32_MAX\" test.\n\n:::note\nThis is the -1 sentinel family, used by the s8/s16 integer buffer-size queries, by the eight SVDF staging queries (arm_svdf_{s8,state_s16_s8,f32,f16}_{input,output}_ctx_get_buffer_size) and by the s8/s16 LSTM temp-buffer queries and the GRU temp queries (arm_lstm_unidirectional_{s8,s16}_temp{1,2}_get_buffer_size, arm_gru_unidirectional_{f32,f16}_temp1_get_buffer_size). The four f32/f16 LSTM temp queries have no dimensions to fold (the buffers are unused) and answer -1 only for NULL params, 0 otherwise. It is not interchangeable with the arm_nn_checked_size_mul() / arm_nn_size_to_i32_or_zero() helpers in Source/NNSupportFunctions (shared header for the float sizers), which most f32 and f16 buffer-size queries use and which report an out-of-range size as 0. Mixing the two silently flips a sizer's out-of-range contract from \"must never be used to size a buffer\" to \"you may pass { NULL, 0 }\", so pick the one the surrounding family already uses.\n\n:::\n\n:::note\nThe split is per sizer, not per datatype. The four SVDF f32/f16 staging queries deliberately use this -1 family rather than the 0 one their neighbours use, because their kernels read ctx->size and a size of 0 opts out of the scratch-size check - so a 0-on-overflow answer fed back as { alloc(0), 0 } would disable the very check meant to catch it. Do not infer a sizer's sentinel from its datatype suffix.\n\n:::",
          "examples": [],
          "id": "arm_nn_size_mul",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_size_mul",
          "params": [
            {
              "description": "Running product, or -1 if an earlier fold already overflowed.",
              "direction": "in",
              "name": "acc",
              "type": "const int64_t"
            },
            {
              "description": "Next factor to fold in.",
              "direction": "in",
              "name": "factor",
              "type": "const int64_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "acc * factor, or -1 if acc is already -1, factor is negative or out of int32_t range, or the product exceeds INT32_MAX."
            }
          ],
          "signature": "static int64_t arm_nn_size_mul(const int64_t acc, const int64_t factor)",
          "source": {
            "line": 307,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L307"
          },
          "summary": "Fold one dimension into a running buffer-size product, reporting overflow as -1."
        },
        {
          "description": "Add to a running buffer-size product, reporting overflow as -1.\n\nCompanion to arm_nn_size_mul() for the sizers that append a fixed slack term.\n\n:::note\nSame sentinel caveat as arm_nn_size_mul() - see its note for the full split, including why the four SVDF f32/f16 staging queries use this -1 family rather than the 0-returning arm_nn_checked_size_mul() / arm_nn_size_to_i32_or_zero() family that most other float sizers use.\n\n:::",
          "examples": [],
          "id": "arm_nn_size_add",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_size_add",
          "params": [
            {
              "description": "Running product, or -1 if an earlier step already overflowed.",
              "direction": "in",
              "name": "acc",
              "type": "const int64_t"
            },
            {
              "description": "Value to add. Must be non-negative.",
              "direction": "in",
              "name": "addend",
              "type": "const int64_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "acc + addend, or -1 if acc is already -1 or the sum exceeds INT32_MAX."
            }
          ],
          "signature": "static int64_t arm_nn_size_add(const int64_t acc, const int64_t addend)",
          "source": {
            "line": 332,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L332"
          },
          "summary": "Add to a running buffer-size product, reporting overflow as -1."
        },
        {
          "description": "definition to pack four 8 bit values.\n\nByte lanes are masked and shifted in uint32_t so a negative value never feeds a signed left shift (UB); masking before the shift keeps the same bits the old shift-then-mask form kept. Bit-identical for every input. Deliberate divergence from upstream ARM-software/CMSIS-NN, which still carries the signed-shift form  do not paste the upstream text back on a sync (issue #357).",
          "examples": [],
          "id": "PACK_S8x4_32x1",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "PACK_S8x4_32x1",
          "params": [
            {
              "description": "",
              "name": "v0"
            },
            {
              "description": "",
              "name": "v1"
            },
            {
              "description": "",
              "name": "v2"
            },
            {
              "description": "",
              "name": "v3"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define PACK_S8x4_32x1(v0, v1, v2, v3) ((int32_t)((((uint32_t)(v0)) & 0xFFu) | ((((uint32_t)(v1)) & 0xFFu) << 8) | ((((uint32_t)(v2)) & 0xFFu) << 16) | \\ ((((uint32_t)(v3)) & 0xFFu) << 24)))",
          "source": {
            "line": 358,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L358"
          },
          "summary": "definition to pack four 8 bit values."
        },
        {
          "description": "definition to pack two 16 bit values.\n\nSame treatment: the high half is shifted in uint32_t, not int32_t, so a negative v1 is defined; the low half keeps its mask. Bit-identical for every input. Same deliberate upstream divergence as PACK_S8x4_32x1 above.",
          "examples": [],
          "id": "PACK_Q15x2_32x1",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "PACK_Q15x2_32x1",
          "params": [
            {
              "description": "",
              "name": "v0"
            },
            {
              "description": "",
              "name": "v1"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define PACK_Q15x2_32x1(v0, v1) ((int32_t)((((uint32_t)(v0)) & 0xFFFFu) | (((uint32_t)(v1)) << 16)))",
          "source": {
            "line": 369,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L369"
          },
          "summary": "definition to pack two 16 bit values."
        },
        {
          "description": "Map an output index to the nearest input index for resize.\n\nThis helper follows the TensorFlow Lite nearest-neighbor resize mapping rules.",
          "examples": [],
          "id": "GetNearestNeighbor",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "GetNearestNeighbor",
          "params": [
            {
              "description": "Output index (x or y).",
              "direction": "in",
              "name": "input_value",
              "type": "const int"
            },
            {
              "description": "Input size along the same axis.",
              "direction": "in",
              "name": "input_size",
              "type": "const int32_t"
            },
            {
              "description": "Precomputed scaling factor for the axis.",
              "direction": "in",
              "name": "scale",
              "type": "const float"
            },
            {
              "description": "Precomputed offset for the axis.",
              "direction": "in",
              "name": "offset",
              "type": "const float"
            },
            {
              "description": "If true, use align-corners scaling.",
              "direction": "in",
              "name": "align_corners",
              "type": "const bool"
            },
            {
              "description": "If true, use half-pixel center offset.",
              "direction": "in",
              "name": "half_pixel_centers",
              "type": "const bool"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Nearest input index for the given output index."
            }
          ],
          "signature": "static int32_t GetNearestNeighbor(\n    const int input_value,\n    const int32_t input_size,\n    const float scale,\n    const float offset,\n    const bool align_corners,\n    const bool half_pixel_centers\n)",
          "source": {
            "line": 391,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L391"
          },
          "summary": "Map an output index to the nearest input index for resize."
        },
        {
          "description": "Check if convolution parameters correspond to a 1x1 convolution.",
          "examples": [],
          "id": "arm_nn_is_convolve_1x1",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_is_convolve_1x1",
          "params": [
            {
              "description": "Convolution parameters",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Input dimensions",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter dimensions",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "true if parameters describe a 1x1 convolution, false otherwise."
            }
          ],
          "signature": "static bool arm_nn_is_convolve_1x1(\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_dims *input_dims,\n    const cmsis_nn_dims *filter_dims\n)",
          "source": {
            "line": 416,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L416"
          },
          "summary": "Check if convolution parameters correspond to a 1x1 convolution."
        },
        {
          "description": "Check if a 1x1 convolution qualifies for the fast (unit stride) path.\n\n:::note\nDoes not validate that the kernel is 1x1. Call arm_nn_is_convolve_1x1() first.\n\n:::",
          "examples": [],
          "id": "arm_nn_is_convolve_1x1_fast",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_is_convolve_1x1_fast",
          "params": [
            {
              "description": "Convolution parameters",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "true if stride is 1x1, false otherwise."
            }
          ],
          "signature": "static bool arm_nn_is_convolve_1x1_fast(const cmsis_nn_conv_params *conv_params)",
          "source": {
            "line": 432,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L432"
          },
          "summary": "Check if a 1x1 convolution qualifies for the fast (unit stride) path."
        },
        {
          "description": "Check if convolution parameters correspond to a 1xN convolution.",
          "examples": [],
          "id": "arm_nn_is_convolve_1_x_n",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_is_convolve_1_x_n",
          "params": [
            {
              "description": "Convolution parameters",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Input dimensions",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter dimensions",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "true if parameters describe a 1xN convolution, false otherwise."
            }
          ],
          "signature": "static bool arm_nn_is_convolve_1_x_n(\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_dims *input_dims,\n    const cmsis_nn_dims *filter_dims\n)",
          "source": {
            "line": 444,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L444"
          },
          "summary": "Check if convolution parameters correspond to a 1xN convolution."
        },
        {
          "description": "Check that `arm_convolve_1_x_n_s4()` handles the horizontal padding of a 1xN convolution.\n\nThe kernel places pad.w columns on the left and pad.w + (total_pad % 2) on the right, where total_pad = (output W - 1) * stride.w + filter W - input W, and needs the output columns that read padding to fit in output W. Its padded-column code also assumes that each such column reads at least one input column and that the filter is no wider than the input; otherwise it forms input and filter addresses outside the tensors. On MVE builds another pad placement, or too many padded columns, returns ARM_CMSIS_NN_FAILURE. A VALID layer whose stride leaves trailing input unused (negative total_pad) is therefore rejected, and the wrapper routes it to another convolution. The kernel also computes a single output row, so vertical padding or an output height other than 1 is rejected. A non-positive stride.w is left to the kernel's argument checks.",
          "examples": [],
          "id": "arm_nn_convolve_1_x_n_padding_supported",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_convolve_1_x_n_padding_supported",
          "params": [
            {
              "description": "Convolution parameters",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Input dimensions",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter dimensions",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Output dimensions",
              "direction": "in",
              "name": "output_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "true when `arm_convolve_1_x_n_s4()` handles the padding, false otherwise."
            }
          ],
          "signature": "static bool arm_nn_convolve_1_x_n_padding_supported(\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_dims *input_dims,\n    const cmsis_nn_dims *filter_dims,\n    const cmsis_nn_dims *output_dims\n)",
          "source": {
            "line": 473,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L473"
          },
          "summary": "Check that armconvolve1xns4() handles the horizontal padding of a 1xN convolution."
        },
        {
          "description": "Check that `arm_convolve_1_x_n_s8()` accepts the padding and output shape of a 1xN convolution.\n\nThe kernel computes a single output row for any pad.w >= 0 and any output width, including an odd total padding, a filter wider than the input and a VALID layer whose stride leaves trailing input unused. It rejects vertical padding, an output height other than 1, a negative pad.w and an empty filter; the wrapper routes those layers to another convolution. A non-positive stride.w is left to the kernel's argument checks.",
          "examples": [],
          "id": "arm_nn_convolve_1_x_n_s8_padding_supported",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_convolve_1_x_n_s8_padding_supported",
          "params": [
            {
              "description": "Convolution parameters",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Filter dimensions",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Output dimensions",
              "direction": "in",
              "name": "output_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "true when `arm_convolve_1_x_n_s8()` computes the layer, false otherwise."
            }
          ],
          "signature": "static bool arm_nn_convolve_1_x_n_s8_padding_supported(\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_dims *filter_dims,\n    const cmsis_nn_dims *output_dims\n)",
          "source": {
            "line": 517,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L517"
          },
          "summary": "Check that armconvolve1xns8() accepts the padding and output shape of a 1xN convolution."
        },
        {
          "description": "Count the output columns of a 1xN convolution whose window reads padding.\n\nOutput column j reads input columns j * stride.w - pad.w to j * stride.w - pad.w + filter W - 1. The leading columns whose window starts before the input are left-padded; of the others, the trailing columns whose window ends past the input are right-padded. A window can do both only when it is left-padded.",
          "examples": [],
          "id": "arm_nn_convolve_1_x_n_padded_columns",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_convolve_1_x_n_padded_columns",
          "params": [
            {
              "description": "Convolution parameters. stride.w >= 1 and pad.w >= 0.",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Input dimensions. w >= 0.",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter dimensions. w >= 1.",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Output dimensions. w >= 0.",
              "direction": "in",
              "name": "output_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Number of left-padded output columns, output W at most.",
              "direction": "out",
              "name": "left_num",
              "type": "int64_t *"
            },
            {
              "description": "Number of right-padded output columns, output W - left_num at most.",
              "direction": "out",
              "name": "right_num",
              "type": "int64_t *"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_nn_convolve_1_x_n_padded_columns(\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_dims *input_dims,\n    const cmsis_nn_dims *filter_dims,\n    const cmsis_nn_dims *output_dims,\n    int64_t *left_num,\n    int64_t *right_num\n)",
          "source": {
            "line": 539,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L539"
          },
          "summary": "Count the output columns of a 1xN convolution whose window reads padding."
        },
        {
          "description": "Check if the dilation, stride and padding of a depthwise layer allow the `arm_depthwise_conv_s8_opt()` or `arm_depthwise_conv_fast_s16()` route.\n\n:::note\nDoes not check ch_mult, the batch count or the kernel size: `arm_depthwise_conv_wrapper_s8()`, `arm_depthwise_conv_wrapper_s16()` and their buffer-size functions apply their own conditions on those, and all of them take this predicate so that routing and sizing agree.\n\n:::",
          "examples": [],
          "id": "arm_nn_dw_conv_opt_dilation_supported",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_dw_conv_opt_dilation_supported",
          "params": [
            {
              "description": "Depthwise convolution parameters",
              "direction": "in",
              "name": "dw_conv_params",
              "type": "const cmsis_nn_dw_conv_params *"
            },
            {
              "description": "Input dimensions",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter dimensions",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Output dimensions",
              "direction": "in",
              "name": "output_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "true for an undilated layer (dilation 1 in both dimensions), or for a 1D layer dilated along the width only: filter, input and output height 1, stride 1 in both dimensions, no vertical padding, dilation.h == 1 and dilation.w >= 1. false otherwise."
            }
          ],
          "signature": "static bool arm_nn_dw_conv_opt_dilation_supported(\n    const cmsis_nn_dw_conv_params *dw_conv_params,\n    const cmsis_nn_dims *input_dims,\n    const cmsis_nn_dims *filter_dims,\n    const cmsis_nn_dims *output_dims\n)",
          "source": {
            "line": 572,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L572"
          },
          "summary": "Check if the dilation, stride and padding of a depthwise layer allow the armdepthwiseconvs8opt() or armdepthwiseconvfasts16() route."
        },
        {
          "description": "Converts the elements from a s8 vector to a s16 vector with an added offset.\n\nOutput elements are ordered. The equation used for the conversion process is:\n\ndst[n] = (int16_t) src[n] + offset; 0 <= n < block_size.",
          "examples": [],
          "id": "arm_q7_to_q15_with_offset",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_q7_to_q15_with_offset",
          "params": [
            {
              "description": "pointer to the s8 input vector",
              "direction": "in",
              "name": "src",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to the s16 output vector",
              "direction": "out",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "length of the input vector",
              "direction": "in",
              "name": "block_size",
              "type": "int32_t"
            },
            {
              "description": "s16 offset to be added to each input vector element.",
              "direction": "in",
              "name": "offset",
              "type": "int16_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_q7_to_q15_with_offset(const int8_t *src, int16_t *dst, int32_t block_size, int16_t offset)",
          "source": {
            "line": 649,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L649"
          },
          "summary": "Converts the elements from a s8 vector to a s16 vector with an added offset."
        },
        {
          "description": "Get the required buffer size for optimized s8 depthwise convolution function with constraint that in_channel equals out_channel. This is for processors with MVE extension.\n\nThe dimensions are checked here, so a negative dimension returns -1 on every build target. The byte count is range-checked inside the selected leg instead, because the Helium and DSP legs use different formulas and the plain-C build needs no buffer at all.\n\n:::note\nIntended for compilation on Host. If compiling for an Arm target, use `arm_depthwise_conv_s8_opt_get_buffer_size()`. Note also this is a support function, so not recommended to call directly even on Host.\n\n:::\n\n:::note\nThis leg sizes its buffer from a fixed channel block rather than from input_dims->c, but it applies the same dimension check as `arm_depthwise_conv_s8_opt_get_buffer_size()` anyway, returning -1 for a negative input_dims->c or filter dimension, or for a byte count that would not fit in an int32_t. Without that check this entry point - and every s4 depthwise sizer, which route here - answered a negative channel count with a plausible positive size (issue #318).\n\n:::",
          "examples": [],
          "id": "arm_depthwise_conv_s8_opt_get_buffer_size_mve",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_depthwise_conv_s8_opt_get_buffer_size_mve",
          "params": [
            {
              "description": "Input (activation) tensor dimensions. Format: [1, H, W, C_IN] Batch argument N is not used.",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter tensor dimensions. Format: [1, H, W, C_OUT]",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns required buffer size in bytes, or -1 if any dimension it reads is negative or the required size would not fit in an int32_t"
            }
          ],
          "signature": "int32_t arm_depthwise_conv_s8_opt_get_buffer_size_mve(const cmsis_nn_dims *input_dims, const cmsis_nn_dims *filter_dims)",
          "source": {
            "line": 695,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L695"
          },
          "summary": "Get the required buffer size for optimized s8 depthwise convolution function with constraint that inchannel equals outchannel."
        },
        {
          "description": "Get the required buffer size for optimized s8 depthwise convolution function with constraint that in_channel equals out_channel. This is for processors with DSP extension.\n\nThe dimensions are checked here, so a negative dimension returns -1 on every build target. The byte count is range-checked inside the selected leg instead, because the Helium and DSP legs use different formulas and the plain-C build needs no buffer at all.\n\n:::note\nIntended for compilation on Host. If compiling for an Arm target, use `arm_depthwise_conv_s8_opt_get_buffer_size()`. Note also this is a support function, so not recommended to call directly even on Host.\n\n:::",
          "examples": [],
          "id": "arm_depthwise_conv_s8_opt_get_buffer_size_dsp",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_depthwise_conv_s8_opt_get_buffer_size_dsp",
          "params": [
            {
              "description": "Input (activation) tensor dimensions. Format: [1, H, W, C_IN] Batch argument N is not used.",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter tensor dimensions. Format: [1, H, W, C_OUT]",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns required buffer size in bytes, or -1 if any dimension it reads is negative or the required size would not fit in an int32_t"
            }
          ],
          "signature": "int32_t arm_depthwise_conv_s8_opt_get_buffer_size_dsp(const cmsis_nn_dims *input_dims, const cmsis_nn_dims *filter_dims)",
          "source": {
            "line": 710,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L710"
          },
          "summary": "Get the required buffer size for optimized s8 depthwise convolution function with constraint that inchannel equals outchannel."
        },
        {
          "description": "Depthwise conv on an im2col buffer where the input channel equals output channel.\n\nSupported framework: TensorFlow Lite micro.",
          "examples": [],
          "id": "arm_nn_depthwise_conv_s8_core",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv_s8_core",
          "params": [
            {
              "description": "pointer to row",
              "direction": "in",
              "name": "row",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to im2col buffer, always consists of 2 columns.",
              "direction": "in",
              "name": "col",
              "type": "const int16_t *"
            },
            {
              "description": "number of channels",
              "direction": "in",
              "name": "num_ch",
              "type": "const uint16_t"
            },
            {
              "description": "pointer to per output channel requantization shift parameter.",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t *"
            },
            {
              "description": "pointer to per output channel requantization multiplier parameter.",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t *"
            },
            {
              "description": "output tensor offset.",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "minimum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "maximum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "number of elements in one column.",
              "direction": "in",
              "name": "kernel_size",
              "type": "const uint16_t"
            },
            {
              "description": "per output channel bias. Range : int32",
              "direction": "in",
              "name": "output_bias",
              "type": "const int32_t *const"
            },
            {
              "description": "pointer to output",
              "direction": "out",
              "name": "out",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns one of the two\n\n1. The incremented output pointer for a successful operation or\n2. NULL if implementation is not available."
            }
          ],
          "signature": "int8_t * arm_nn_depthwise_conv_s8_core(\n    const int8_t *row,\n    const int16_t *col,\n    const uint16_t num_ch,\n    const int32_t *out_shift,\n    const int32_t *out_mult,\n    const int32_t out_offset,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const uint16_t kernel_size,\n    const int32_t *const output_bias,\n    int8_t *out\n)",
          "source": {
            "line": 732,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L732"
          },
          "summary": "Depthwise conv on an im2col buffer where the input channel equals output channel."
        },
        {
          "description": "General Matrix-multiplication function with per-channel requantization.\n\nSupported framework: TensorFlow Lite",
          "examples": [],
          "id": "arm_nn_mat_mult_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_s8",
          "params": [
            {
              "description": "pointer to row operand",
              "direction": "in",
              "name": "input_row",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to col operand",
              "direction": "in",
              "name": "input_col",
              "type": "const int8_t *"
            },
            {
              "description": "number of rows of input_row",
              "direction": "in",
              "name": "output_ch",
              "type": "const uint16_t"
            },
            {
              "description": "number of column batches. Range: 1 to 4",
              "direction": "in",
              "name": "col_batches",
              "type": "const uint16_t"
            },
            {
              "description": "pointer to per output channel requantization shift parameter.",
              "direction": "in",
              "name": "output_shift",
              "type": "const int32_t *"
            },
            {
              "description": "pointer to per output channel requantization multiplier parameter.",
              "direction": "in",
              "name": "output_mult",
              "type": "const int32_t *"
            },
            {
              "description": "output tensor offset.",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "input tensor(col) offset.",
              "direction": "in",
              "name": "col_offset",
              "type": "const int32_t"
            },
            {
              "description": "kernel offset(row). Not used.",
              "direction": "in",
              "name": "row_offset",
              "type": "const int32_t"
            },
            {
              "description": "minimum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "out_activation_min",
              "type": "const int16_t"
            },
            {
              "description": "maximum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "out_activation_max",
              "type": "const int16_t"
            },
            {
              "description": "number of elements in each row",
              "direction": "in",
              "name": "row_len",
              "type": "const uint16_t"
            },
            {
              "description": "per output channel bias. Range : int32",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *const"
            },
            {
              "description": "pointer to output",
              "direction": "inout",
              "name": "out",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns one of the two\n\n1. The incremented output pointer for a successful operation or\n2. NULL if implementation is not available."
            }
          ],
          "signature": "int8_t * arm_nn_mat_mult_s8(\n    const int8_t *input_row,\n    const int8_t *input_col,\n    const uint16_t output_ch,\n    const uint16_t col_batches,\n    const int32_t *output_shift,\n    const int32_t *output_mult,\n    const int32_t out_offset,\n    const int32_t col_offset,\n    const int32_t row_offset,\n    const int16_t out_activation_min,\n    const int16_t out_activation_max,\n    const uint16_t row_len,\n    const int32_t *const bias,\n    int8_t *out\n)",
          "source": {
            "line": 766,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L766"
          },
          "summary": "General Matrix-multiplication function with per-channel requantization."
        },
        {
          "description": "Matrix-multiplication function for convolution with per-channel requantization for 16 bits convolution.\n\nThis function does the matrix multiplication of weight matrix for all output channels with 2 columns from im2col and produces two elements/output_channel. The outputs are clamped in the range provided by activation min and max. Supported framework: TensorFlow Lite micro.",
          "examples": [],
          "id": "arm_nn_mat_mult_kernel_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_kernel_s16",
          "params": [
            {
              "description": "pointer to operand A",
              "direction": "in",
              "name": "input_a",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to operand B, always consists of 2 vectors.",
              "direction": "in",
              "name": "input_b",
              "type": "const int16_t *"
            },
            {
              "description": "number of rows of A",
              "direction": "in",
              "name": "output_ch",
              "type": "const int32_t"
            },
            {
              "description": "pointer to per output channel requantization shift parameter.",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t *"
            },
            {
              "description": "pointer to per output channel requantization multiplier parameter.",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t *"
            },
            {
              "description": "minimum value to clamp the output to. Range : int16",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "maximum value to clamp the output to. Range : int16",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "number of columns of A",
              "direction": "in",
              "name": "num_col_a",
              "type": "const int32_t"
            },
            {
              "description": "pointer to struct with bias vector. The length of this vector is equal to the number of output columns (or RHS input rows). The vector can be int32 or int64 indicated by a flag in the struct.",
              "direction": "in",
              "name": "bias_data",
              "type": "const cmsis_nn_bias_data *const"
            },
            {
              "description": "pointer to output",
              "direction": "inout",
              "name": "out_0",
              "type": "int16_t *"
            },
            {
              "description": "Address offset between rows in output.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns one of the two\n\n1. The incremented output pointer for a successful operation or\n2. NULL if implementation is not available."
            }
          ],
          "signature": "int16_t * arm_nn_mat_mult_kernel_s16(\n    const int8_t *input_a,\n    const int16_t *input_b,\n    const int32_t output_ch,\n    const int32_t *out_shift,\n    const int32_t *out_mult,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const int32_t num_col_a,\n    const cmsis_nn_bias_data *const bias_data,\n    int16_t *out_0,\n    const int32_t row_address_offset\n)",
          "source": {
            "line": 804,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L804"
          },
          "summary": "Matrix-multiplication function for convolution with per-channel requantization for 16 bits convolution."
        },
        {
          "description": "General Vector by Matrix multiplication with requantization and storage of result.\n\nPseudo-code *output = 0 sum_col = 0 for (j = 0; j < out_ch; j++) for (i = 0; i < row_elements; i++) *output += row_base_ref[i] * col_base_ref[i] sum_col += col_base_ref[i] scale sum_col using quant_params and bias store result in 'output'",
          "examples": [],
          "id": "arm_nn_mat_mul_core_1x_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mul_core_1x_s8",
          "params": [
            {
              "description": "number of row elements",
              "direction": "in",
              "name": "row_elements",
              "type": "int32_t"
            },
            {
              "description": "number of row elements skipped due to padding. row_elements + skipped_row_elements = (kernel_x * kernel_y) * input_ch",
              "direction": "in",
              "name": "skipped_row_elements",
              "type": "const int32_t"
            },
            {
              "description": "pointer to row operand",
              "direction": "in",
              "name": "row_base_ref",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to col operand",
              "direction": "in",
              "name": "col_base_ref",
              "type": "const int8_t *"
            },
            {
              "description": "Number of output channels",
              "direction": "in",
              "name": "out_ch",
              "type": "const int32_t"
            },
            {
              "description": "Pointer to convolution parameters like offsets and activation values",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Pointer to per-channel quantization parameters",
              "direction": "in",
              "name": "quant_params",
              "type": "const cmsis_nn_per_channel_quant_params *"
            },
            {
              "description": "Pointer to optional per-channel bias",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to output where int8 results are stored.",
              "direction": "out",
              "name": "output",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function performs matrix(row_base_ref) multiplication with vector(col_base_ref) and scaled result is stored in memory."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mul_core_1x_s8(\n    int32_t row_elements,\n    const int32_t skipped_row_elements,\n    const int8_t *row_base_ref,\n    const int8_t *col_base_ref,\n    const int32_t out_ch,\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_per_channel_quant_params *quant_params,\n    const int32_t *bias,\n    int8_t *output\n)",
          "source": {
            "line": 843,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L843"
          },
          "summary": "General Vector by Matrix multiplication with requantization and storage of result."
        },
        {
          "description": "General Vector by Matrix multiplication with requantization, storage of result and int4 weights packed into an int8 buffer.\n\nPseudo-code as int8 example. Int4 filter data will be unpacked. *output = 0 sum_col = 0 for (j = 0; j < out_ch; j++) for (i = 0; i < row_elements; i++) *output += row_base_ref[i] * col_base_ref[i] sum_col += col_base_ref[i] scale sum_col using quant_params and bias store result in 'output'",
          "examples": [],
          "id": "arm_nn_mat_mul_core_1x_s4",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mul_core_1x_s4",
          "params": [
            {
              "description": "number of row elements",
              "direction": "in",
              "name": "row_elements",
              "type": "int32_t"
            },
            {
              "description": "number of row elements skipped due to padding. row_elements + skipped_row_elements = (kernel_x * kernel_y) * input_ch",
              "direction": "in",
              "name": "skipped_row_elements",
              "type": "const int32_t"
            },
            {
              "description": "pointer to row operand",
              "direction": "in",
              "name": "row_base_ref",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to col operand as packed int4",
              "direction": "in",
              "name": "col_base_ref",
              "type": "const int8_t *"
            },
            {
              "description": "Number of output channels",
              "direction": "in",
              "name": "out_ch",
              "type": "const int32_t"
            },
            {
              "description": "Pointer to convolution parameters like offsets and activation values",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Pointer to per-channel quantization parameters",
              "direction": "in",
              "name": "quant_params",
              "type": "const cmsis_nn_per_channel_quant_params *"
            },
            {
              "description": "Pointer to optional per-channel bias",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to output where int8 results are stored.",
              "direction": "out",
              "name": "output",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function performs matrix(row_base_ref) multiplication with vector(col_base_ref) and scaled result is stored in memory."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mul_core_1x_s4(\n    int32_t row_elements,\n    const int32_t skipped_row_elements,\n    const int8_t *row_base_ref,\n    const int8_t *col_base_ref,\n    const int32_t out_ch,\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_per_channel_quant_params *quant_params,\n    const int32_t *bias,\n    int8_t *output\n)",
          "source": {
            "line": 881,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L881"
          },
          "summary": "General Vector by Matrix multiplication with requantization, storage of result and int4 weights packed into an int8 buffer."
        },
        {
          "description": "Matrix-multiplication with requantization & activation function for four rows and one column.\n\nCompliant to TFLM int8 specification. MVE implementation only",
          "examples": [],
          "id": "arm_nn_mat_mul_core_4x_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mul_core_4x_s8",
          "params": [
            {
              "description": "number of row elements",
              "direction": "in",
              "name": "row_elements",
              "type": "const int32_t"
            },
            {
              "description": "offset between rows. Can be the same as row_elements. For e.g, in a 1x1 conv scenario with stride as 1.",
              "direction": "in",
              "name": "offset",
              "type": "const int32_t"
            },
            {
              "description": "pointer to row operand",
              "direction": "in",
              "name": "row_base",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to col operand",
              "direction": "in",
              "name": "col_base",
              "type": "const int8_t *"
            },
            {
              "description": "Number of output channels",
              "direction": "in",
              "name": "out_ch",
              "type": "const int32_t"
            },
            {
              "description": "Pointer to convolution parameters like offsets and activation values",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Pointer to per-channel quantization parameters",
              "direction": "in",
              "name": "quant_params",
              "type": "const cmsis_nn_per_channel_quant_params *"
            },
            {
              "description": "Pointer to per-channel bias",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to output where int8 results are stored.",
              "direction": "out",
              "name": "output",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns the updated output pointer or NULL if implementation is not available."
            }
          ],
          "signature": "int8_t * arm_nn_mat_mul_core_4x_s8(\n    const int32_t row_elements,\n    const int32_t offset,\n    const int8_t *row_base,\n    const int8_t *col_base,\n    const int32_t out_ch,\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_per_channel_quant_params *quant_params,\n    const int32_t *bias,\n    int8_t *output\n)",
          "source": {
            "line": 908,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L908"
          },
          "summary": "Matrix-multiplication with requantization & activation function for four rows and one column."
        },
        {
          "description": "General Matrix-multiplication function with per-channel requantization. This function assumes:\n\n- LHS input matrix NOT transposed (nt)\n- RHS input matrix transposed (t)\n- RHS is int8 packed with 2x int4\n- LHS is int8\n\n:::note\nThis operation also performs the broadcast bias addition before the requantization\n\n:::",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_t_s4",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_t_s4",
          "params": [
            {
              "description": "Pointer to the LHS input matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the RHS input matrix",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the bias vector. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the output matrix with \"m\" rows and \"n\" columns",
              "direction": "out",
              "name": "dst",
              "type": "int8_t *"
            },
            {
              "description": "Pointer to the multipliers vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_multipliers",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the shifts vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_shifts",
              "type": "const int32_t *"
            },
            {
              "description": "Number of LHS input rows",
              "direction": "in",
              "name": "lhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of RHS input rows",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of LHS/RHS input columns",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be applied to the LHS input value",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be applied the output result",
              "direction": "in",
              "name": "dst_offset",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp down the output. Range : int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp up the output. Range : int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "Column offset between subsequent lhs_rows",
              "direction": "in",
              "name": "lhs_cols_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_t_s4(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t *bias,\n    int8_t *dst,\n    const int32_t *dst_multipliers,\n    const int32_t *dst_shifts,\n    const int32_t lhs_rows,\n    const int32_t rhs_rows,\n    const int32_t rhs_cols,\n    const int32_t lhs_offset,\n    const int32_t dst_offset,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const int32_t lhs_cols_offset\n)",
          "source": {
            "line": 950,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L950"
          },
          "summary": "General Matrix-multiplication function with per-channel requantization."
        },
        {
          "description": "General Matrix-multiplication function with per-channel requantization. This function assumes:\n\n- LHS input matrix NOT transposed (nt)\n- RHS input matrix transposed (t)\n- RHS is int8 packed with 2x int4\n- LHS is int8\n- LHS/RHS input columns must be even numbered\n- LHS must be interleaved. Compare to arm_nn_mat_mult_nt_t_s4 where LHS is not interleaved.\n\n:::note\nThis operation also performs the broadcast bias addition before the requantization\n\n:::",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_interleaved_t_even_s4",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_interleaved_t_even_s4",
          "params": [
            {
              "description": "Pointer to the LHS input matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the RHS input matrix",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the bias vector. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the output matrix with \"m\" rows and \"n\" columns",
              "direction": "out",
              "name": "dst",
              "type": "int8_t *"
            },
            {
              "description": "Pointer to the multipliers vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_multipliers",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the shifts vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_shifts",
              "type": "const int32_t *"
            },
            {
              "description": "Number of LHS input rows",
              "direction": "in",
              "name": "lhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of RHS input rows",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of LHS/RHS input columns. Note this must be even.",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be applied to the LHS input value",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be applied the output result",
              "direction": "in",
              "name": "dst_offset",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp down the output. Range : int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp up the output. Range : int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "Column offset between subsequent lhs_rows",
              "direction": "in",
              "name": "lhs_cols_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_interleaved_t_even_s4(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t *bias,\n    int8_t *dst,\n    const int32_t *dst_multipliers,\n    const int32_t *dst_shifts,\n    const int32_t lhs_rows,\n    const int32_t rhs_rows,\n    const int32_t rhs_cols,\n    const int32_t lhs_offset,\n    const int32_t dst_offset,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const int32_t lhs_cols_offset\n)",
          "source": {
            "line": 999,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L999"
          },
          "summary": "General Matrix-multiplication function with per-channel requantization."
        },
        {
          "description": "General Matrix-multiplication function with per-channel requantization. This function assumes:\n\n- LHS input matrix NOT transposed (nt)\n- RHS input matrix transposed (t)\n\n:::note\nThis operation also performs the broadcast bias addition before the requantization\n\n:::",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_t_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_t_s8",
          "params": [
            {
              "description": "Pointer to the weight sum multiplied by lhs_offset and summed bias buffer",
              "direction": "in",
              "name": "weight_sum_buf",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the LHS input matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the RHS input matrix",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the bias vector. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the output matrix with \"m\" rows and \"n\" columns",
              "direction": "out",
              "name": "dst",
              "type": "int8_t *"
            },
            {
              "description": "Pointer to the multipliers vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_multipliers",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the shifts vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_shifts",
              "type": "const int32_t *"
            },
            {
              "description": "Number of LHS input rows",
              "direction": "in",
              "name": "lhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of RHS input rows",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of LHS/RHS input columns",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be applied to the LHS input value",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be applied the output result",
              "direction": "in",
              "name": "dst_offset",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp down the output. Range : int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp up the output. Range : int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "Address offset between rows in output. NOTE: Only used for MVEI extension.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "const int32_t"
            },
            {
              "description": "Column offset between subsequent lhs_rows",
              "direction": "in",
              "name": "lhs_cols_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_t_s8(\n    const int32_t *weight_sum_buf,\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t *bias,\n    int8_t *dst,\n    const int32_t *dst_multipliers,\n    const int32_t *dst_shifts,\n    const int32_t lhs_rows,\n    const int32_t rhs_rows,\n    const int32_t rhs_cols,\n    const int32_t lhs_offset,\n    const int32_t dst_offset,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const int32_t row_address_offset,\n    const int32_t lhs_cols_offset\n)",
          "source": {
            "line": 1046,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1046"
          },
          "summary": "General Matrix-multiplication function with per-channel requantization."
        },
        {
          "description": "General Matrix-multiplication function with per-channel requantization. Output is calculated with multiple channels in parallel, rather than multiple output indices in a single channel This function assumes:\n\n- LHS input matrix NOT transposed (nt)\n- RHS input matrix transposed (t)\n\n:::note\nThis operation also performs the broadcast bias addition before the requantization\n\n:::",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_t_1x1_out_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_t_1x1_out_s8",
          "params": [
            {
              "description": "Pointer to the weight sum multiplied by lhs_offset and summed bias buffer",
              "direction": "in",
              "name": "weight_sum_buf",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the LHS input matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the RHS input matrix",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the bias vector. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the output matrix with \"m\" rows and \"n\" columns",
              "direction": "out",
              "name": "dst",
              "type": "int8_t *"
            },
            {
              "description": "Pointer to the multipliers vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_multipliers",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the shifts vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_shifts",
              "type": "const int32_t *"
            },
            {
              "description": "Number of LHS input rows",
              "direction": "in",
              "name": "lhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of RHS input rows",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of LHS/RHS input columns",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be applied to the LHS input value",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be applied the output result",
              "direction": "in",
              "name": "dst_offset",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp down the output. Range : int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp up the output. Range : int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "Address offset between rows in output. NOTE: Only used for MVEI extension.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "const int32_t"
            },
            {
              "description": "Column offset between subsequent lhs_rows",
              "direction": "in",
              "name": "lhs_cols_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_t_1x1_out_s8(\n    const int32_t *weight_sum_buf,\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t *bias,\n    int8_t *dst,\n    const int32_t *dst_multipliers,\n    const int32_t *dst_shifts,\n    const int32_t lhs_rows,\n    const int32_t rhs_rows,\n    const int32_t rhs_cols,\n    const int32_t lhs_offset,\n    const int32_t dst_offset,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const int32_t row_address_offset,\n    const int32_t lhs_cols_offset\n)",
          "source": {
            "line": 1097,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1097"
          },
          "summary": "General Matrix-multiplication function with per-channel requantization."
        },
        {
          "description": "General Matrix-multiplication function with per-channel requantization and int16 input (LHS) and output. This function assumes:\n\n- LHS input matrix NOT transposed (nt)\n- RHS input matrix transposed (t)\n\n:::note\nThis operation also performs the broadcast bias addition before the requantization\n\n:::\n\nMVE implementation only.",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_t_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_t_s16",
          "params": [
            {
              "description": "Pointer to the LHS input matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int16_t *"
            },
            {
              "description": "Pointer to the RHS input matrix",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to struct with bias vector. The length of this vector is equal to the number of output columns (or RHS input rows). The vector can be int32 or int64 indicated by a flag in the struct.",
              "direction": "in",
              "name": "bias_data",
              "type": "const cmsis_nn_bias_data *"
            },
            {
              "description": "Pointer to the output matrix with \"m\" rows and \"n\" columns",
              "direction": "out",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "Pointer to the multipliers vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_multipliers",
              "type": "const int32_t *"
            },
            {
              "description": "Pointer to the shifts vector needed for the per-channel requantization. The length of this vector is equal to the number of output columns (or RHS input rows)",
              "direction": "in",
              "name": "dst_shifts",
              "type": "const int32_t *"
            },
            {
              "description": "Number of LHS input rows",
              "direction": "in",
              "name": "lhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of RHS input rows",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of LHS/RHS input columns",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp down the output. Range : int16",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp up the output. Range : int16",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "Address offset between rows in output. NOTE: Only used for MVEI extension.",
              "direction": "in",
              "name": "row_address_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS` or `ARM_CMSIS_NN_NO_IMPL_ERROR` if not for MVE |---row_address_offset---| |____rhs_rows__________________|\n\n|  |  |\n| --- | --- |\n|  |  |\n|  |  |\n\n| | | lhs_rows\n\n|  |  |\n| --- | --- |\n| _______________ | ______________ |"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_t_s16(\n    const int16_t *lhs,\n    const int8_t *rhs,\n    const cmsis_nn_bias_data *bias_data,\n    int16_t *dst,\n    const int32_t *dst_multipliers,\n    const int32_t *dst_shifts,\n    const int32_t lhs_rows,\n    const int32_t rhs_rows,\n    const int32_t rhs_cols,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const int32_t row_address_offset\n)",
          "source": {
            "line": 1155,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1155"
          },
          "summary": "General Matrix-multiplication function with per-channel requantization and int16 input (LHS) and output."
        },
        {
          "description": "General Matrix-multiplication function with int8 input and int32 output. This function assumes:\n\n- LHS input matrix NOT transposed (nt)\n- RHS input matrix transposed (t)\n\n:::note\nDst/output buffer must be zeroed out before calling this function.\n\n:::",
          "examples": [],
          "id": "arm_nn_mat_mult_nt_t_s8_s32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_nt_t_s8_s32",
          "params": [
            {
              "description": "Pointer to the LHS input matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the RHS input matrix",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Pointer to the output matrix with \"m\" rows and \"n\" columns. Accumulated into, so it must be zeroed by the caller before the call",
              "direction": "inout",
              "name": "dst",
              "type": "int32_t *"
            },
            {
              "description": "Number of LHS input rows",
              "direction": "in",
              "name": "lhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of LHS input columns/RHS input rows",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of RHS input columns",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be applied to the LHS input value",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset between subsequent output results",
              "direction": "in",
              "name": "dst_idx_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_mat_mult_nt_t_s8_s32(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    int32_t *dst,\n    const int32_t lhs_rows,\n    const int32_t rhs_rows,\n    const int32_t rhs_cols,\n    const int32_t lhs_offset,\n    const int32_t dst_idx_offset\n)",
          "source": {
            "line": 1189,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1189"
          },
          "summary": "General Matrix-multiplication function with int8 input and int32 output."
        },
        {
          "description": "s4 Vector by Matrix (transposed) multiplication",
          "examples": [],
          "id": "arm_nn_vec_mat_mult_t_s4",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_vec_mat_mult_t_s4",
          "params": [
            {
              "description": "Input left-hand side vector",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "packed_rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input bias",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Output vector",
              "direction": "out",
              "name": "dst",
              "type": "int8_t *"
            },
            {
              "description": "Offset to be added to the input values of the left-hand side vector. Range: -127 to 128",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be added to the output values. Range: -127 to 128",
              "direction": "in",
              "name": "dst_offset",
              "type": "const int32_t"
            },
            {
              "description": "Output multiplier",
              "direction": "in",
              "name": "dst_multiplier",
              "type": "const int32_t"
            },
            {
              "description": "Output shift",
              "direction": "in",
              "name": "dst_shift",
              "type": "const int32_t"
            },
            {
              "description": "Number of columns in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Number of rows in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_vec_mat_mult_t_s4(\n    const int8_t *lhs,\n    const int8_t *packed_rhs,\n    const int32_t *bias,\n    int8_t *dst,\n    const int32_t lhs_offset,\n    const int32_t dst_offset,\n    const int32_t dst_multiplier,\n    const int32_t dst_shift,\n    const int32_t rhs_cols,\n    const int32_t rhs_rows,\n    const int32_t activation_min,\n    const int32_t activation_max\n)",
          "source": {
            "line": 1218,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1218"
          },
          "summary": "s4 Vector by Matrix (transposed) multiplication"
        },
        {
          "description": "s8 Vector by Matrix (transposed) multiplication",
          "examples": [],
          "id": "arm_nn_vec_mat_mult_t_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_vec_mat_mult_t_s8",
          "params": [
            {
              "description": "Input left-hand side vector",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Kernel sums of the kernels (rhs). See arm_vector_sum_s8 for more info.",
              "direction": "in",
              "name": "kernel_sum",
              "type": "const int32_t *"
            },
            {
              "description": "Input bias",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Output vector",
              "direction": "out",
              "name": "dst",
              "type": "int8_t *"
            },
            {
              "description": "Offset to be added to the input values of the left-hand side vector. Range: -127 to 128",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be added to the output values. Range: -127 to 128",
              "direction": "in",
              "name": "dst_offset",
              "type": "const int32_t"
            },
            {
              "description": "Output multiplier",
              "direction": "in",
              "name": "dst_multiplier",
              "type": "const int32_t"
            },
            {
              "description": "Output shift",
              "direction": "in",
              "name": "dst_shift",
              "type": "const int32_t"
            },
            {
              "description": "Number of columns in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Number of rows in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "Memory position offset for dst. First output is stored at 'dst', the second at 'dst + address_offset' and so on. Default value is typically 1.",
              "direction": "in",
              "name": "address_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be added to the input values of the right-hand side vector. Range: -127 to 128",
              "direction": "in",
              "name": "rhs_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_vec_mat_mult_t_s8(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t *kernel_sum,\n    const int32_t *bias,\n    int8_t *dst,\n    const int32_t lhs_offset,\n    const int32_t dst_offset,\n    const int32_t dst_multiplier,\n    const int32_t dst_shift,\n    const int32_t rhs_cols,\n    const int32_t rhs_rows,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const int32_t address_offset,\n    const int32_t rhs_offset\n)",
          "source": {
            "line": 1256,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1256"
          },
          "summary": "s8 Vector by Matrix (transposed) multiplication"
        },
        {
          "description": "s8 Vector by Matrix (transposed) multiplication using per channel quantization for output",
          "examples": [],
          "id": "arm_nn_vec_mat_mult_t_per_ch_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_vec_mat_mult_t_per_ch_s8",
          "params": [
            {
              "description": "Input left-hand side vector",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Kernel sums of the kernels (rhs). See arm_vector_sum_s8 for more info.",
              "direction": "in",
              "name": "kernel_sum",
              "type": "const int32_t *"
            },
            {
              "description": "Input bias",
              "direction": "in",
              "name": "bias",
              "type": "const int32_t *"
            },
            {
              "description": "Output vector",
              "direction": "out",
              "name": "dst",
              "type": "int8_t *"
            },
            {
              "description": "Offset to be added to the input values of the left-hand side vector. Range: -127 to 128",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be added to the output values. Range: -127 to 128",
              "direction": "in",
              "name": "dst_offset",
              "type": "const int32_t"
            },
            {
              "description": "Output multipliers",
              "direction": "in",
              "name": "dst_multiplier",
              "type": "const int32_t *"
            },
            {
              "description": "Output shifts",
              "direction": "in",
              "name": "dst_shift",
              "type": "const int32_t *"
            },
            {
              "description": "Number of columns in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Number of rows in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "Memory position offset for dst. First output is stored at 'dst', the second at 'dst + address_offset' and so on. Default value is typically 1.",
              "direction": "in",
              "name": "address_offset",
              "type": "const int32_t"
            },
            {
              "description": "Offset to be added to the input values of the right-hand side vector. Range: -127 to 128",
              "direction": "in",
              "name": "rhs_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_vec_mat_mult_t_per_ch_s8(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t *kernel_sum,\n    const int32_t *bias,\n    int8_t *dst,\n    const int32_t lhs_offset,\n    const int32_t dst_offset,\n    const int32_t *dst_multiplier,\n    const int32_t *dst_shift,\n    const int32_t rhs_cols,\n    const int32_t rhs_rows,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const int32_t address_offset,\n    const int32_t rhs_offset\n)",
          "source": {
            "line": 1297,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1297"
          },
          "summary": "s8 Vector by Matrix (transposed) multiplication using per channel quantization for output"
        },
        {
          "description": "s16 Vector by s8 Matrix (transposed) multiplication",
          "examples": [],
          "id": "arm_nn_vec_mat_mult_t_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_vec_mat_mult_t_s16",
          "params": [
            {
              "description": "Input left-hand side vector",
              "direction": "in",
              "name": "lhs",
              "type": "const int16_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input bias",
              "direction": "in",
              "name": "bias",
              "type": "const int64_t *"
            },
            {
              "description": "Output vector",
              "direction": "out",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "Output multiplier",
              "direction": "in",
              "name": "dst_multiplier",
              "type": "const int32_t"
            },
            {
              "description": "Output shift",
              "direction": "in",
              "name": "dst_shift",
              "type": "const int32_t"
            },
            {
              "description": "Number of columns in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Number of rows in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int16",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int16",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_vec_mat_mult_t_s16(\n    const int16_t *lhs,\n    const int8_t *rhs,\n    const int64_t *bias,\n    int16_t *dst,\n    const int32_t dst_multiplier,\n    const int32_t dst_shift,\n    const int32_t rhs_cols,\n    const int32_t rhs_rows,\n    const int32_t activation_min,\n    const int32_t activation_max\n)",
          "source": {
            "line": 1330,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1330"
          },
          "summary": "s16 Vector by s8 Matrix (transposed) multiplication"
        },
        {
          "description": "s16 vector(lhs) by s8 matrix (transposed) multiplication and per channel quant output",
          "examples": [],
          "id": "arm_nn_vec_mat_mult_t_per_ch_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_vec_mat_mult_t_per_ch_s16",
          "params": [
            {
              "description": "Input left-hand side vector",
              "direction": "in",
              "name": "lhs",
              "type": "const int16_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input bias",
              "direction": "in",
              "name": "bias",
              "type": "const int64_t *"
            },
            {
              "description": "Output vector",
              "direction": "out",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "Per channel output multiplier. Length of vector is equal to rhs_rows",
              "direction": "in",
              "name": "dst_multiplier",
              "type": "const int32_t *"
            },
            {
              "description": "Per channel output shift. Length of vector is equal to rhs_rows",
              "direction": "in",
              "name": "dst_shift",
              "type": "const int32_t *"
            },
            {
              "description": "Number of columns in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Number of rows in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int16",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int16",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_vec_mat_mult_t_per_ch_s16(\n    const int16_t *lhs,\n    const int8_t *rhs,\n    const int64_t *bias,\n    int16_t *dst,\n    const int32_t *dst_multiplier,\n    const int32_t *dst_shift,\n    const int32_t rhs_cols,\n    const int32_t rhs_rows,\n    const int32_t activation_min,\n    const int32_t activation_max\n)",
          "source": {
            "line": 1358,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1358"
          },
          "summary": "s16 vector(lhs) by s8 matrix (transposed) multiplication and per channel quant output"
        },
        {
          "description": "s16 Vector by s16 Matrix (transposed) multiplication",
          "examples": [],
          "id": "arm_nn_vec_mat_mult_t_s16_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_vec_mat_mult_t_s16_s16",
          "params": [
            {
              "description": "Input left-hand side vector",
              "direction": "in",
              "name": "lhs",
              "type": "const int16_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "rhs",
              "type": "const int16_t *"
            },
            {
              "description": "Input bias",
              "direction": "in",
              "name": "bias",
              "type": "const int64_t *"
            },
            {
              "description": "Output vector",
              "direction": "out",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "Output multiplier",
              "direction": "in",
              "name": "dst_multiplier",
              "type": "const int32_t"
            },
            {
              "description": "Output shift",
              "direction": "in",
              "name": "dst_shift",
              "type": "const int32_t"
            },
            {
              "description": "Number of columns in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Number of rows in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int16",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int16",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_vec_mat_mult_t_s16_s16(\n    const int16_t *lhs,\n    const int16_t *rhs,\n    const int64_t *bias,\n    int16_t *dst,\n    const int32_t dst_multiplier,\n    const int32_t dst_shift,\n    const int32_t rhs_cols,\n    const int32_t rhs_rows,\n    const int32_t activation_min,\n    const int32_t activation_max\n)",
          "source": {
            "line": 1386,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1386"
          },
          "summary": "s16 Vector by s16 Matrix (transposed) multiplication"
        },
        {
          "description": "s8 Vector by Matrix (transposed) multiplication with s16 output",
          "examples": [],
          "id": "arm_nn_vec_mat_mult_t_svdf_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_vec_mat_mult_t_svdf_s8",
          "params": [
            {
              "description": "Input left-hand side vector",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Output vector",
              "direction": "out",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "Offset to be added to the input values of the left-hand side vector. Range: -127 to 128",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Address offset for dst. First output is stored at 'dst', the second at 'dst + scatter_offset' and so on.",
              "direction": "in",
              "name": "scatter_offset",
              "type": "const int32_t"
            },
            {
              "description": "Output multiplier",
              "direction": "in",
              "name": "dst_multiplier",
              "type": "const int32_t"
            },
            {
              "description": "Output shift",
              "direction": "in",
              "name": "dst_shift",
              "type": "const int32_t"
            },
            {
              "description": "Number of columns in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Number of rows in the right-hand side input matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int16",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int16",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_vec_mat_mult_t_svdf_s8(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    int16_t *dst,\n    const int32_t lhs_offset,\n    const int32_t scatter_offset,\n    const int32_t dst_multiplier,\n    const int32_t dst_shift,\n    const int32_t rhs_cols,\n    const int32_t rhs_rows,\n    const int32_t activation_min,\n    const int32_t activation_max\n)",
          "source": {
            "line": 1417,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1417"
          },
          "summary": "s8 Vector by Matrix (transposed) multiplication with s16 output"
        },
        {
          "description": "Depthwise convolution of transposed rhs matrix with 4 lhs matrices. To be used in padded cases where the padding is -lhs_offset(Range: int8). Dimensions are the same for lhs and rhs.\n\n:::note\nTail channel loads and stores are predicated, so channel-indexed arrays are not accessed beyond `active_ch`.\n\n:::",
          "examples": [],
          "id": "arm_nn_depthwise_conv_nt_t_padded_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv_nt_t_padded_s8",
          "params": [
            {
              "description": "Input left-hand side matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "LHS matrix offset(input offset). Range: -127 to 128",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Subset of total_ch processed",
              "direction": "in",
              "name": "active_ch",
              "type": "const int32_t"
            },
            {
              "description": "Number of channels in LHS/RHS",
              "direction": "in",
              "name": "total_ch",
              "type": "const int32_t"
            },
            {
              "description": "Per channel output shift. Length of vector is equal to number of channels",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t *"
            },
            {
              "description": "Per channel output multiplier. Length of vector is equal to number of channels",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t *"
            },
            {
              "description": "Offset to be added to the output values. Range: -127 to 128",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "(row_dimension * col_dimension) of LHS/RHS matrix",
              "direction": "in",
              "name": "row_x_col",
              "type": "const uint16_t"
            },
            {
              "description": "Per channel output bias. Length of vector is equal to number of channels",
              "direction": "in",
              "name": "output_bias",
              "type": "const int32_t *const"
            },
            {
              "description": "Output pointer",
              "direction": "out",
              "name": "out",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS` if an implementation is available or `ARM_CMSIS_NN_NO_IMPL_ERROR` otherwise"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_depthwise_conv_nt_t_padded_s8(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t lhs_offset,\n    const int32_t active_ch,\n    const int32_t total_ch,\n    const int32_t *out_shift,\n    const int32_t *out_mult,\n    const int32_t out_offset,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const uint16_t row_x_col,\n    const int32_t *const output_bias,\n    int8_t *out\n)",
          "source": {
            "line": 1453,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1453"
          },
          "summary": "Depthwise convolution of transposed rhs matrix with 4 lhs matrices."
        },
        {
          "description": "Depthwise convolution of transposed rhs matrix with 4 lhs matrices. To be used in non-padded cases. Dimensions are the same for lhs and rhs.\n\n:::note\nTail channel loads and stores are predicated, so channel-indexed arrays are not accessed beyond `active_ch`.\n\n:::",
          "examples": [],
          "id": "arm_nn_depthwise_conv_nt_t_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv_nt_t_s8",
          "params": [
            {
              "description": "Pointer to the weight sum multiplied by lhs_offset and summed bias buffer",
              "direction": "in",
              "name": "weight_sum_buf",
              "type": "const int32_t *"
            },
            {
              "description": "Input left-hand side matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "LHS matrix offset(input offset). Range: -127 to 128",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Subset of total_ch processed",
              "direction": "in",
              "name": "active_ch",
              "type": "const int32_t"
            },
            {
              "description": "Number of channels in LHS/RHS",
              "direction": "in",
              "name": "total_ch",
              "type": "const int32_t"
            },
            {
              "description": "Per channel output shift. Length of vector is equal to number of channels.",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t *"
            },
            {
              "description": "Per channel output multiplier. Length of vector is equal to number of channels.",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t *"
            },
            {
              "description": "Offset to be added to the output values. Range: -127 to 128",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "(row_dimension * col_dimension) of LHS/RHS matrix",
              "direction": "in",
              "name": "row_x_col",
              "type": "const uint16_t"
            },
            {
              "description": "Per channel output bias. Length of vector is equal to number of channels.",
              "direction": "in",
              "name": "output_bias",
              "type": "const int32_t *const"
            },
            {
              "description": "Output pointer",
              "direction": "out",
              "name": "out",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS` if an implementation is available or `ARM_CMSIS_NN_NO_IMPL_ERROR` otherwise"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_depthwise_conv_nt_t_s8(\n    const int32_t *weight_sum_buf,\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t lhs_offset,\n    const int32_t active_ch,\n    const int32_t total_ch,\n    const int32_t *out_shift,\n    const int32_t *out_mult,\n    const int32_t out_offset,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const uint16_t row_x_col,\n    const int32_t *const output_bias,\n    int8_t *out\n)",
          "source": {
            "line": 1492,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1492"
          },
          "summary": "Depthwise convolution of transposed rhs matrix with 4 lhs matrices."
        },
        {
          "description": "Necessary conditions of the planar rule that are cheap to test inline: at most 32 channels and stride 1. A caller can skip `arm_nn_depthwise_conv_s8_planar()` for layers that fail them without changing which layers it takes.",
          "examples": [],
          "id": "arm_nn_depthwise_conv_s8_planar_candidate",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv_s8_planar_candidate",
          "params": [
            {
              "description": "Depthwise convolution parameters",
              "direction": "in",
              "name": "dw_conv_params",
              "type": "const cmsis_nn_dw_conv_params *"
            },
            {
              "description": "Input tensor dimensions. Format: [1, H, W, C_IN]",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "1 when the layer may take the planar path, 0 when it cannot."
            }
          ],
          "signature": "static int32_t arm_nn_depthwise_conv_s8_planar_candidate(\n    const cmsis_nn_dw_conv_params *dw_conv_params,\n    const cmsis_nn_dims *input_dims\n)",
          "source": {
            "line": 1517,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1517"
          },
          "summary": "Necessary conditions of the planar rule that are cheap to test inline: at most 32 channels and stride 1."
        },
        {
          "description": "The gate of `arm_convolve_s8_small_cin()`: upscale_dims NULL, input depth 1 to 3 with filter depth equal to it, dilation 1, a kernel of at least 1x1 with kernel width x depth at most 16 and at most 48 values, and a positive multiple of 4 output channels. Plain C; it evaluates the same on every build.",
          "examples": [],
          "id": "arm_nn_is_convolve_s8_small_cin",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_is_convolve_s8_small_cin",
          "params": [
            {
              "description": "Convolution parameters",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Input tensor dimensions. Format: [N, H, W, C_IN]",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter tensor dimensions. Format: [C_OUT, HK, WK, CK]",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Output tensor dimensions. Format: [N, H, W, C_OUT]",
              "direction": "in",
              "name": "output_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Upscale tensor dimensions, or NULL",
              "direction": "in",
              "name": "upscale_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "1 when the layer is in the gate, 0 otherwise."
            }
          ],
          "signature": "static int32_t arm_nn_is_convolve_s8_small_cin(\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_dims *input_dims,\n    const cmsis_nn_dims *filter_dims,\n    const cmsis_nn_dims *output_dims,\n    const cmsis_nn_dims *upscale_dims\n)",
          "source": {
            "line": 1536,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1536"
          },
          "summary": "The gate of armconvolves8smallcin(): upscaledims NULL, input depth 1 to 3 with filter depth equal to it, dilation 1, a kernel of at least 1x1 with kernel width…"
        },
        {
          "description": "The gate of `arm_convolve_s8_3x3_c16_s1()`: upscale_dims NULL, input and filter depth 16, a 3x3 kernel, and stride and dilation 1. Plain C; it evaluates the same on every build.",
          "examples": [],
          "id": "arm_nn_is_convolve_s8_3x3_c16_s1",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_is_convolve_s8_3x3_c16_s1",
          "params": [
            {
              "description": "Convolution parameters",
              "direction": "in",
              "name": "conv_params",
              "type": "const cmsis_nn_conv_params *"
            },
            {
              "description": "Input tensor dimensions. Format: [N, H, W, C_IN]",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter tensor dimensions. Format: [C_OUT, HK, WK, CK]",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Upscale tensor dimensions, or NULL",
              "direction": "in",
              "name": "upscale_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "1 when the layer is in the gate, 0 otherwise."
            }
          ],
          "signature": "static int32_t arm_nn_is_convolve_s8_3x3_c16_s1(\n    const cmsis_nn_conv_params *conv_params,\n    const cmsis_nn_dims *input_dims,\n    const cmsis_nn_dims *filter_dims,\n    const cmsis_nn_dims *upscale_dims\n)",
          "source": {
            "line": 1562,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1562"
          },
          "summary": "The gate of armconvolves83x3c16s1(): upscaledims NULL, input and filter depth 16, a 3x3 kernel, and stride and dilation 1."
        },
        {
          "description": "The group check of `arm_convolve_s8()`, for its direct entries: with groups = C_IN / filter C, C_IN or C_OUT is not a multiple of groups. A filter C of zero or above C_IN gives no group count and is not reported.",
          "examples": [],
          "id": "arm_nn_convolve_s8_groups_invalid",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_convolve_s8_groups_invalid",
          "params": [
            {
              "description": "Input tensor dimensions. Format: [N, H, W, C_IN]",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter tensor dimensions. Format: [C_OUT, HK, WK, CK]",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Output tensor dimensions. Format: [N, H, W, C_OUT]",
              "direction": "in",
              "name": "output_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "1 when `arm_convolve_s8()` reports the group count as an argument error, 0 otherwise."
            }
          ],
          "signature": "static int32_t arm_nn_convolve_s8_groups_invalid(\n    const cmsis_nn_dims *input_dims,\n    const cmsis_nn_dims *filter_dims,\n    const cmsis_nn_dims *output_dims\n)",
          "source": {
            "line": 1582,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1582"
          },
          "summary": "The group check of armconvolves8(), for its direct entries: with groups = CIN / filter C, CIN or COUT is not a multiple of groups."
        },
        {
          "description": "Plane size in bytes that `arm_nn_depthwise_conv_s8_planar()` needs for a layer, or -1 when the layer is not one it takes. The rule is plain C and evaluates the same on every build.",
          "examples": [],
          "id": "arm_nn_depthwise_conv_s8_planar_bytes",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv_s8_planar_bytes",
          "params": [
            {
              "description": "Depthwise convolution parameters",
              "direction": "in",
              "name": "dw_conv_params",
              "type": "const cmsis_nn_dw_conv_params *"
            },
            {
              "description": "Input tensor dimensions. Format: [1, H, W, C_IN]",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter tensor dimensions. Format: [1, H, W, C_OUT]",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Output tensor dimensions. Format: [1, H, W, C_OUT]",
              "direction": "in",
              "name": "output_dims",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The plane size in bytes, or -1."
            }
          ],
          "signature": "int32_t arm_nn_depthwise_conv_s8_planar_bytes(\n    const cmsis_nn_dw_conv_params *dw_conv_params,\n    const cmsis_nn_dims *input_dims,\n    const cmsis_nn_dims *filter_dims,\n    const cmsis_nn_dims *output_dims\n)",
          "source": {
            "line": 1601,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1601"
          },
          "summary": "Plane size in bytes that armnndepthwiseconvs8planar() needs for a layer, or -1 when the layer is not one it takes."
        },
        {
          "description": "s8 depthwise convolution with channel multiplier 1 and stride 1, vectorized across the output pixels of one channel plane instead of across channels. It serves the few-channel and 1xk layers of `arm_depthwise_conv_s8_opt()`, with the same scratch buffer and weight sums.",
          "examples": [],
          "id": "arm_nn_depthwise_conv_s8_planar",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv_s8_planar",
          "params": [
            {
              "description": "Scratch buffer of `arm_depthwise_conv_s8_opt_get_buffer_size()` bytes",
              "direction": "inout",
              "name": "ctx",
              "type": "const cmsis_nn_context *"
            },
            {
              "description": "Per-channel weight sums from `arm_depthwise_convolve_weight_sum()`, bias included",
              "direction": "in",
              "name": "weight_sum_ctx",
              "type": "const cmsis_nn_context *"
            },
            {
              "description": "Depthwise convolution parameters",
              "direction": "in",
              "name": "dw_conv_params",
              "type": "const cmsis_nn_dw_conv_params *"
            },
            {
              "description": "Per-channel quantization parameters",
              "direction": "in",
              "name": "quant_params",
              "type": "const cmsis_nn_per_channel_quant_params *"
            },
            {
              "description": "Input tensor dimensions. Format: [1, H, W, C_IN]",
              "direction": "in",
              "name": "input_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Input data pointer",
              "direction": "in",
              "name": "input",
              "type": "const int8_t *"
            },
            {
              "description": "Filter tensor dimensions. Format: [1, H, W, C_OUT]",
              "direction": "in",
              "name": "filter_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Filter data pointer",
              "direction": "in",
              "name": "kernel",
              "type": "const int8_t *"
            },
            {
              "description": "Output tensor dimensions. Format: [1, H, W, C_OUT]",
              "direction": "in",
              "name": "output_dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Output data pointer",
              "direction": "out",
              "name": "output",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "`ARM_CMSIS_NN_SUCCESS` when the layer was computed, or `ARM_CMSIS_NN_NO_IMPL_ERROR` when it is not one this path takes or its plane does not fit in ctx->size (then nothing is written), or MVE is not available."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_depthwise_conv_s8_planar(\n    const cmsis_nn_context *ctx,\n    const cmsis_nn_context *weight_sum_ctx,\n    const cmsis_nn_dw_conv_params *dw_conv_params,\n    const cmsis_nn_per_channel_quant_params *quant_params,\n    const cmsis_nn_dims *input_dims,\n    const int8_t *input,\n    const cmsis_nn_dims *filter_dims,\n    const int8_t *kernel,\n    const cmsis_nn_dims *output_dims,\n    int8_t *output\n)",
          "source": {
            "line": 1626,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1626"
          },
          "summary": "s8 depthwise convolution with channel multiplier 1 and stride 1, vectorized across the output pixels of one channel plane instead of across channels."
        },
        {
          "description": "Depthwise convolution of transposed rhs matrix with 4 lhs matrices. To be used in non-padded cases. rhs consists of packed int4 data. Dimensions are the same for lhs and rhs.\n\n:::note\nTail channel loads and stores are predicated, so channel-indexed arrays are not accessed beyond `active_ch`.\n\n:::",
          "examples": [],
          "id": "arm_nn_depthwise_conv_nt_t_s4",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv_nt_t_s4",
          "params": [
            {
              "description": "Input left-hand side matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed). Consists of int4 data packed in an int8 buffer.",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "LHS matrix offset(input offset). Range: -127 to 128",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Subset of total_ch processed",
              "direction": "in",
              "name": "active_ch",
              "type": "const int32_t"
            },
            {
              "description": "Number of channels in LHS/RHS",
              "direction": "in",
              "name": "total_ch",
              "type": "const int32_t"
            },
            {
              "description": "Per channel output shift. Length of vector is equal to number of channels.",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t *"
            },
            {
              "description": "Per channel output multiplier. Length of vector is equal to number of channels.",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t *"
            },
            {
              "description": "Offset to be added to the output values. Range: -127 to 128",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "(row_dimension * col_dimension) of LHS/RHS matrix",
              "direction": "in",
              "name": "row_x_col",
              "type": "const uint16_t"
            },
            {
              "description": "Per channel output bias. Length of vector is equal to number of channels.",
              "direction": "in",
              "name": "output_bias",
              "type": "const int32_t *const"
            },
            {
              "description": "Output pointer",
              "direction": "out",
              "name": "out",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns one of the two\n\n- Updated output pointer if an implementation is available\n- NULL if no implementation is available."
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_depthwise_conv_nt_t_s4(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t lhs_offset,\n    const int32_t active_ch,\n    const int32_t total_ch,\n    const int32_t *out_shift,\n    const int32_t *out_mult,\n    const int32_t out_offset,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const uint16_t row_x_col,\n    const int32_t *const output_bias,\n    int8_t *out\n)",
          "source": {
            "line": 1663,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1663"
          },
          "summary": "Depthwise convolution of transposed rhs matrix with 4 lhs matrices."
        },
        {
          "description": "Depthwise convolution of transposed rhs matrix with 4 lhs matrices. To be used in non-padded cases. Dimensions are the same for lhs and rhs.\n\n:::note\nTail channel loads and stores are predicated, so channel-indexed arrays are not accessed beyond `num_ch`.\n\n:::",
          "examples": [],
          "id": "arm_nn_depthwise_conv_nt_t_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_depthwise_conv_nt_t_s16",
          "params": [
            {
              "description": "Input left-hand side matrix",
              "direction": "in",
              "name": "lhs",
              "type": "const int16_t *"
            },
            {
              "description": "Input right-hand side matrix (transposed)",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Number of channels in LHS/RHS",
              "direction": "in",
              "name": "num_ch",
              "type": "const uint16_t"
            },
            {
              "description": "Per channel output shift. Length of vector is equal to number of channels.",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t *"
            },
            {
              "description": "Per channel output multiplier. Length of vector is equal to number of channels.",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t *"
            },
            {
              "description": "Minimum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "Maximum value to clamp the output to. Range: int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "(row_dimension * col_dimension) of LHS/RHS matrix",
              "direction": "in",
              "name": "row_x_col",
              "type": "const uint16_t"
            },
            {
              "description": "Per channel output bias. Length of vector is equal to number of channels.",
              "direction": "in",
              "name": "output_bias",
              "type": "const int64_t *const"
            },
            {
              "description": "Output pointer",
              "direction": "out",
              "name": "out",
              "type": "int16_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns one of the two\n\n- Updated output pointer if an implementation is available\n- NULL if no implementation is available."
            }
          ],
          "signature": "int16_t * arm_nn_depthwise_conv_nt_t_s16(\n    const int16_t *lhs,\n    const int8_t *rhs,\n    const uint16_t num_ch,\n    const int32_t *out_shift,\n    const int32_t *out_mult,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const uint16_t row_x_col,\n    const int64_t *const output_bias,\n    int16_t *out\n)",
          "source": {
            "line": 1699,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1699"
          },
          "summary": "Depthwise convolution of transposed rhs matrix with 4 lhs matrices."
        },
        {
          "description": "Row of s8 scalars multiplicated with a s8 matrix ad accumulated into a s32 rolling scratch buffer. Helpfunction for transposed convolution.\n\n:::note\nRolling buffer refers to how the function wraps around the scratch buffer, e.g. it starts writing at [output_start + output_index], writes to [output_start + output_max] and then continues at [output_start] again.\n\n:::",
          "examples": [],
          "id": "arm_nn_transpose_conv_row_s8_s32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_transpose_conv_row_s8_s32",
          "params": [
            {
              "description": "Input left-hand side scalars",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Input right-hand side matrix",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Output buffer start",
              "direction": "out",
              "name": "output_start",
              "type": "int32_t *"
            },
            {
              "description": "Output buffer current index",
              "direction": "in",
              "name": "output_index",
              "type": "const int32_t"
            },
            {
              "description": "Output buffer size",
              "direction": "in",
              "name": "output_max",
              "type": "const int32_t"
            },
            {
              "description": "Number of rows in rhs matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of columns in rhs matrix",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Number of input channels",
              "direction": "in",
              "name": "input_channels",
              "type": "const int32_t"
            },
            {
              "description": "Number of output channels",
              "direction": "in",
              "name": "output_channels",
              "type": "const int32_t"
            },
            {
              "description": "Offset added to lhs before multiplication",
              "direction": "in",
              "name": "lhs_offset",
              "type": "const int32_t"
            },
            {
              "description": "Address offset between each row of data output",
              "direction": "in",
              "name": "row_offset",
              "type": "const int32_t"
            },
            {
              "description": "Length of lhs scalar row.",
              "direction": "in",
              "name": "input_x",
              "type": "const int32_t"
            },
            {
              "description": "Address offset between each scalar-matrix multiplication result.",
              "direction": "in",
              "name": "stride_x",
              "type": "const int32_t"
            },
            {
              "description": "Skip rows on top of the filter, used for padding.",
              "direction": "in",
              "name": "skip_row_top",
              "type": "const int32_t"
            },
            {
              "description": "Skip rows in the bottom of the filter, used for padding.",
              "direction": "in",
              "name": "skip_row_bottom",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns ARM_CMSIS_NN_SUCCESS"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_transpose_conv_row_s8_s32(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    int32_t *output_start,\n    const int32_t output_index,\n    const int32_t output_max,\n    const int32_t rhs_rows,\n    const int32_t rhs_cols,\n    const int32_t input_channels,\n    const int32_t output_channels,\n    const int32_t lhs_offset,\n    const int32_t row_offset,\n    const int32_t input_x,\n    const int32_t stride_x,\n    const int32_t skip_row_top,\n    const int32_t skip_row_bottom\n)",
          "source": {
            "line": 1735,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1735"
          },
          "summary": "Row of s8 scalars multiplicated with a s8 matrix ad accumulated into a s32 rolling scratch buffer."
        },
        {
          "description": "Read 2 s16 elements and post increment pointer.",
          "examples": [],
          "id": "arm_nn_read_q15x2_ia",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_read_q15x2_ia",
          "params": [
            {
              "description": "Pointer to pointer that holds address of input. Advanced past the elements read.",
              "direction": "inout",
              "name": "in_q15",
              "type": "const int16_t **"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "q31 value"
            }
          ],
          "signature": "static int32_t arm_nn_read_q15x2_ia(const int16_t **in_q15)",
          "source": {
            "line": 1756,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1756"
          },
          "summary": "Read 2 s16 elements and post increment pointer."
        },
        {
          "description": "Read 4 s8 from s8 pointer and post increment pointer.",
          "examples": [],
          "id": "arm_nn_read_s8x4_ia",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_read_s8x4_ia",
          "params": [
            {
              "description": "Pointer to pointer that holds address of input. Advanced past the elements read.",
              "direction": "inout",
              "name": "in_s8",
              "type": "const int8_t **"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "q31 value"
            }
          ],
          "signature": "static int32_t arm_nn_read_s8x4_ia(const int8_t **in_s8)",
          "source": {
            "line": 1771,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1771"
          },
          "summary": "Read 4 s8 from s8 pointer and post increment pointer."
        },
        {
          "description": "Read 2 s8 from s8 pointer and post increment pointer.",
          "examples": [],
          "id": "arm_nn_read_s8x2_ia",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_read_s8x2_ia",
          "params": [
            {
              "description": "Pointer to pointer that holds address of input. Advanced past the elements read.",
              "direction": "inout",
              "name": "in_s8",
              "type": "const int8_t **"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "q31 value"
            }
          ],
          "signature": "static int32_t arm_nn_read_s8x2_ia(const int8_t **in_s8)",
          "source": {
            "line": 1785,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1785"
          },
          "summary": "Read 2 s8 from s8 pointer and post increment pointer."
        },
        {
          "description": "Read 2 int16 values from int16 pointer.",
          "examples": [],
          "id": "arm_nn_read_s16x2",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_read_s16x2",
          "params": [
            {
              "description": "pointer to address of input.",
              "direction": "in",
              "name": "in",
              "type": "const int16_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "s32 value"
            }
          ],
          "signature": "static int32_t arm_nn_read_s16x2(const int16_t *in)",
          "source": {
            "line": 1799,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1799"
          },
          "summary": "Read 2 int16 values from int16 pointer."
        },
        {
          "description": "Read 4 s8 values.",
          "examples": [],
          "id": "arm_nn_read_s8x4",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_read_s8x4",
          "params": [
            {
              "description": "pointer to address of input.",
              "direction": "in",
              "name": "in_s8",
              "type": "const int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "s32 value"
            }
          ],
          "signature": "static int32_t arm_nn_read_s8x4(const int8_t *in_s8)",
          "source": {
            "line": 1812,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1812"
          },
          "summary": "Read 4 s8 values."
        },
        {
          "description": "Read 2 s8 values.",
          "examples": [],
          "id": "arm_nn_read_s8x2",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_read_s8x2",
          "params": [
            {
              "description": "pointer to address of input.",
              "direction": "in",
              "name": "in_s8",
              "type": "const int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "s32 value"
            }
          ],
          "signature": "static int32_t arm_nn_read_s8x2(const int8_t *in_s8)",
          "source": {
            "line": 1824,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1824"
          },
          "summary": "Read 2 s8 values."
        },
        {
          "description": "Write four s8 to s8 pointer and increment pointer afterwards.",
          "examples": [],
          "id": "arm_nn_write_s8x4_ia",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_write_s8x4_ia",
          "params": [
            {
              "description": "Double pointer to destination. Advanced past the bytes written.",
              "direction": "inout",
              "name": "in",
              "type": "int8_t **"
            },
            {
              "description": "Four bytes to copy",
              "direction": "in",
              "name": "value",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_nn_write_s8x4_ia(int8_t **in, int32_t value)",
          "source": {
            "line": 1837,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1837"
          },
          "summary": "Write four s8 to s8 pointer and increment pointer afterwards."
        },
        {
          "description": "memset optimized for MVE",
          "examples": [],
          "id": "arm_memset_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memset_s8",
          "params": [
            {
              "description": "Destination pointer",
              "direction": "inout",
              "name": "dst",
              "type": "int8_t *"
            },
            {
              "description": "Value to set",
              "direction": "in",
              "name": "val",
              "type": "const int8_t"
            },
            {
              "description": "Number of bytes to copy.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memset_s8(int8_t *dst, const int8_t val, uint32_t block_size)",
          "source": {
            "line": 1850,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1850"
          },
          "summary": "memset optimized for MVE"
        },
        {
          "description": "memset optimized for MVE for 16-bit data.",
          "examples": [],
          "id": "arm_memset_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memset_s16",
          "params": [
            {
              "description": "Destination pointer.",
              "direction": "inout",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "16-bit value to set.",
              "direction": "in",
              "name": "val",
              "type": "const int16_t"
            },
            {
              "description": "Number of int16_t values to set.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memset_s16(int16_t *dst, const int16_t val, uint32_t block_size)",
          "source": {
            "line": 1873,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L1873"
          },
          "summary": "memset optimized for MVE for 16-bit data."
        },
        {
          "description": "Matrix-multiplication function for convolution with per-channel requantization and 4 bit weights.\n\nThis function does the matrix multiplication of weight matrix for all output channels with 2 columns from im2col and produces two elements/output_channel. The outputs are clamped in the range provided by activation min and max. Supported framework: TensorFlow Lite micro.",
          "examples": [],
          "id": "arm_nn_mat_mult_kernel_s4_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_kernel_s4_s16",
          "params": [
            {
              "description": "pointer to operand A, int8 packed with 2x int4.",
              "direction": "in",
              "name": "input_a",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to operand B, always consists of 2 vectors.",
              "direction": "in",
              "name": "input_b",
              "type": "const int16_t *"
            },
            {
              "description": "number of rows of A",
              "direction": "in",
              "name": "output_ch",
              "type": "const uint16_t"
            },
            {
              "description": "pointer to per output channel requantization shift parameter.",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t *"
            },
            {
              "description": "pointer to per output channel requantization multiplier parameter.",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t *"
            },
            {
              "description": "output tensor offset.",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "minimum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int32_t"
            },
            {
              "description": "maximum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int32_t"
            },
            {
              "description": "number of columns of A",
              "direction": "in",
              "name": "num_col_a",
              "type": "const int32_t"
            },
            {
              "description": "per output channel bias. Range : int32",
              "direction": "in",
              "name": "output_bias",
              "type": "const int32_t *const"
            },
            {
              "description": "pointer to output",
              "direction": "inout",
              "name": "out_0",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns one of the two\n\n1. The incremented output pointer for a successful operation or\n2. NULL if implementation is not available."
            }
          ],
          "signature": "int8_t * arm_nn_mat_mult_kernel_s4_s16(\n    const int8_t *input_a,\n    const int16_t *input_b,\n    const uint16_t output_ch,\n    const int32_t *out_shift,\n    const int32_t *out_mult,\n    const int32_t out_offset,\n    const int32_t activation_min,\n    const int32_t activation_max,\n    const int32_t num_col_a,\n    const int32_t *const output_bias,\n    int8_t *out_0\n)",
          "source": {
            "line": 2094,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2094"
          },
          "summary": "Matrix-multiplication function for convolution with per-channel requantization and 4 bit weights."
        },
        {
          "description": "Matrix-multiplication function for convolution with per-channel requantization.\n\nThis function does the matrix multiplication of weight matrix for all output channels with 2 columns from im2col and produces two elements/output_channel. The outputs are clamped in the range provided by activation min and max. Supported framework: TensorFlow Lite micro.",
          "examples": [],
          "id": "arm_nn_mat_mult_kernel_s8_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_kernel_s8_s16",
          "params": [
            {
              "description": "pointer to operand A",
              "direction": "in",
              "name": "input_a",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to operand B, always consists of 2 vectors.",
              "direction": "in",
              "name": "input_b",
              "type": "const int16_t *"
            },
            {
              "description": "number of rows of A",
              "direction": "in",
              "name": "output_ch",
              "type": "const uint16_t"
            },
            {
              "description": "pointer to per output channel requantization shift parameter.",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t *"
            },
            {
              "description": "pointer to per output channel requantization multiplier parameter.",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t *"
            },
            {
              "description": "output tensor offset.",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "minimum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int16_t"
            },
            {
              "description": "maximum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int16_t"
            },
            {
              "description": "number of columns of A",
              "direction": "in",
              "name": "num_col_a",
              "type": "const int32_t"
            },
            {
              "description": "number of columns of A aligned by 4",
              "direction": "in",
              "name": "aligned_num_col_a",
              "type": "const int32_t"
            },
            {
              "description": "per output channel bias. Range : int32",
              "direction": "in",
              "name": "output_bias",
              "type": "const int32_t *const"
            },
            {
              "description": "pointer to output",
              "direction": "inout",
              "name": "out_0",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns one of the two\n\n1. The incremented output pointer for a successful operation or\n2. NULL if implementation is not available."
            }
          ],
          "signature": "int8_t * arm_nn_mat_mult_kernel_s8_s16(\n    const int8_t *input_a,\n    const int16_t *input_b,\n    const uint16_t output_ch,\n    const int32_t *out_shift,\n    const int32_t *out_mult,\n    const int32_t out_offset,\n    const int16_t activation_min,\n    const int16_t activation_max,\n    const int32_t num_col_a,\n    const int32_t aligned_num_col_a,\n    const int32_t *const output_bias,\n    int8_t *out_0\n)",
          "source": {
            "line": 2128,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2128"
          },
          "summary": "Matrix-multiplication function for convolution with per-channel requantization."
        },
        {
          "description": "Matrix-multiplication function for convolution with per-channel requantization, supporting an address offset between rows.\n\nThis function does the matrix multiplication of weight matrix for all output channels with 2 columns from im2col and produces two elements/output_channel. The outputs are clamped in the range provided by activation min and max.\n\nThis function is slighly less performant than arm_nn_mat_mult_kernel_s8_s16, but allows support for grouped convolution. Supported framework: TensorFlow Lite micro.",
          "examples": [],
          "id": "arm_nn_mat_mult_kernel_row_offset_s8_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mat_mult_kernel_row_offset_s8_s16",
          "params": [
            {
              "description": "pointer to operand A",
              "direction": "in",
              "name": "input_a",
              "type": "const int8_t *"
            },
            {
              "description": "pointer to operand B, always consists of 2 vectors.",
              "direction": "in",
              "name": "input_b",
              "type": "const int16_t *"
            },
            {
              "description": "number of rows of A",
              "direction": "in",
              "name": "output_ch",
              "type": "const uint16_t"
            },
            {
              "description": "pointer to per output channel requantization shift parameter.",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t *"
            },
            {
              "description": "pointer to per output channel requantization multiplier parameter.",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t *"
            },
            {
              "description": "output tensor offset.",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "minimum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "activation_min",
              "type": "const int16_t"
            },
            {
              "description": "maximum value to clamp the output to. Range : int8",
              "direction": "in",
              "name": "activation_max",
              "type": "const int16_t"
            },
            {
              "description": "number of columns of A",
              "direction": "in",
              "name": "num_col_a",
              "type": "const int32_t"
            },
            {
              "description": "number of columns of A aligned by 4",
              "direction": "in",
              "name": "aligned_num_col_a",
              "type": "const int32_t"
            },
            {
              "description": "per output channel bias. Range : int32",
              "direction": "in",
              "name": "output_bias",
              "type": "const int32_t *const"
            },
            {
              "description": "address offset between rows in the output",
              "direction": "in",
              "name": "row_address_offset",
              "type": "const int32_t"
            },
            {
              "description": "pointer to output",
              "direction": "inout",
              "name": "out_0",
              "type": "int8_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns one of the two\n\n1. The incremented output pointer for a successful operation or\n2. NULL if implementation is not available."
            }
          ],
          "signature": "int8_t * arm_nn_mat_mult_kernel_row_offset_s8_s16(\n    const int8_t *input_a,\n    const int16_t *input_b,\n    const uint16_t output_ch,\n    const int32_t *out_shift,\n    const int32_t *out_mult,\n    const int32_t out_offset,\n    const int16_t activation_min,\n    const int16_t activation_max,\n    const int32_t num_col_a,\n    const int32_t aligned_num_col_a,\n    const int32_t *const output_bias,\n    const int32_t row_address_offset,\n    int8_t *out_0\n)",
          "source": {
            "line": 2168,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2168"
          },
          "summary": "Matrix-multiplication function for convolution with per-channel requantization, supporting an address offset between rows."
        },
        {
          "description": "Common softmax function for s8 input and s8 or s16 output.\n\n:::note\nSupported framework: TensorFlow Lite micro (bit-accurate)\n\n:::",
          "examples": [],
          "id": "arm_nn_softmax_common_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_softmax_common_s8",
          "params": [
            {
              "description": "Pointer to the input tensor",
              "direction": "in",
              "name": "input",
              "type": "const int8_t *"
            },
            {
              "description": "Number of rows in the input tensor",
              "direction": "in",
              "name": "num_rows",
              "type": "const int32_t"
            },
            {
              "description": "Number of elements in each input row",
              "direction": "in",
              "name": "row_size",
              "type": "const int32_t"
            },
            {
              "description": "Input quantization multiplier",
              "direction": "in",
              "name": "mult",
              "type": "const int32_t"
            },
            {
              "description": "Input quantization shift within the range [0, 31]",
              "direction": "in",
              "name": "shift",
              "type": "const int32_t"
            },
            {
              "description": "Minimum difference with max in row. Used to check if the quantized exponential operation can be performed",
              "direction": "in",
              "name": "diff_min",
              "type": "const int32_t"
            },
            {
              "description": "Indicating s8 output if 0 else s16 output",
              "direction": "in",
              "name": "int16_output",
              "type": "const bool"
            },
            {
              "description": "Pointer to the output tensor",
              "direction": "out",
              "name": "output",
              "type": "void *"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "void arm_nn_softmax_common_s8(\n    const int8_t *input,\n    const int32_t num_rows,\n    const int32_t row_size,\n    const int32_t mult,\n    const int32_t shift,\n    const int32_t diff_min,\n    const bool int16_output,\n    void *output\n)",
          "source": {
            "line": 2197,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2197"
          },
          "summary": "Common softmax function for s8 input and s8 or s16 output."
        },
        {
          "description": "macro for adding rounding offset",
          "examples": [],
          "id": "NN_ROUND",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "NN_ROUND",
          "params": [
            {
              "description": "",
              "name": "out_shift"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define NN_ROUND(out_shift) ((0x1 << out_shift) >> 1)",
          "source": {
            "line": 2210,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2210"
          },
          "summary": "macro for adding rounding offset"
        },
        {
          "description": "",
          "examples": [],
          "id": "MUL_SAT",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "MUL_SAT",
          "params": [
            {
              "description": "",
              "name": "a"
            },
            {
              "description": "",
              "name": "b"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define MUL_SAT(a, b) arm_nn_doubling_high_mult((a), (b))",
          "source": {
            "line": 2216,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2216"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "MUL_SAT_MVE",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "MUL_SAT_MVE",
          "params": [
            {
              "description": "",
              "name": "a"
            },
            {
              "description": "",
              "name": "b"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define MUL_SAT_MVE(a, b) arm_doubling_high_mult_mve_32x4((a), (b))",
          "source": {
            "line": 2217,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2217"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "MUL_POW2",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "MUL_POW2",
          "params": [
            {
              "description": "",
              "name": "a"
            },
            {
              "description": "",
              "name": "b"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define MUL_POW2(a, b) arm_nn_mult_by_power_of_two((a), (b))",
          "source": {
            "line": 2218,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2218"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "DIV_POW2",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "DIV_POW2",
          "params": [
            {
              "description": "",
              "name": "a"
            },
            {
              "description": "",
              "name": "b"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define DIV_POW2(a, b) arm_nn_divide_by_power_of_two((a), (b))",
          "source": {
            "line": 2220,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2220"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "DIV_POW2_MVE",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "DIV_POW2_MVE",
          "params": [
            {
              "description": "",
              "name": "a"
            },
            {
              "description": "",
              "name": "b"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define DIV_POW2_MVE(a, b) arm_divide_by_power_of_two_mve((a), (b))",
          "source": {
            "line": 2221,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2221"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "EXP_ON_NEG",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "EXP_ON_NEG",
          "params": [
            {
              "description": "",
              "name": "x"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define EXP_ON_NEG(x) arm_nn_exp_on_negative_values((x))",
          "source": {
            "line": 2223,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2223"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "ONE_OVER1",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ONE_OVER1",
          "params": [
            {
              "description": "",
              "name": "x"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define ONE_OVER1(x) arm_nn_one_over_one_plus_x_for_x_in_0_1((x))",
          "source": {
            "line": 2224,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2224"
          },
          "summary": ""
        },
        {
          "description": "Saturating doubling high multiply. Result matches NEON instruction VQRDMULH.",
          "examples": [],
          "id": "arm_nn_doubling_high_mult",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_doubling_high_mult",
          "params": [
            {
              "description": "Multiplicand. Range: {NN_Q31_MIN, NN_Q31_MAX}",
              "direction": "in",
              "name": "m1",
              "type": "const int32_t"
            },
            {
              "description": "Multiplier. Range: {NN_Q31_MIN, NN_Q31_MAX}",
              "direction": "in",
              "name": "m2",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Result of multiplication."
            }
          ],
          "signature": "static int32_t arm_nn_doubling_high_mult(const int32_t m1, const int32_t m2)",
          "source": {
            "line": 2234,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2234"
          },
          "summary": "Saturating doubling high multiply."
        },
        {
          "description": "Doubling high multiply without saturation. This is intended for requantization where the scale is a positive integer.\n\n:::note\nThe result of this matches that of neon instruction VQRDMULH for m1 in range {NN_Q31_MIN, NN_Q31_MAX} and m2 in range {NN_Q31_MIN + 1, NN_Q31_MAX}. Saturation occurs when m1 equals m2 equals NN_Q31_MIN and that is not handled by this function.\n\n:::",
          "examples": [],
          "id": "arm_nn_doubling_high_mult_no_sat",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_doubling_high_mult_no_sat",
          "params": [
            {
              "description": "Multiplicand. Range: {NN_Q31_MIN, NN_Q31_MAX}",
              "direction": "in",
              "name": "m1",
              "type": "int32_t"
            },
            {
              "description": "Multiplier Range: {NN_Q31_MIN, NN_Q31_MAX}",
              "direction": "in",
              "name": "m2",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Result of multiplication."
            }
          ],
          "signature": "static int32_t arm_nn_doubling_high_mult_no_sat(int32_t m1, int32_t m2)",
          "source": {
            "line": 2272,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2272"
          },
          "summary": "Doubling high multiply without saturation."
        },
        {
          "description": "Rounding divide by power of two.",
          "examples": [],
          "id": "arm_nn_divide_by_power_of_two",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_divide_by_power_of_two",
          "params": [
            {
              "description": "- Dividend",
              "direction": "in",
              "name": "dividend",
              "type": "const int32_t"
            },
            {
              "description": "- Divisor = power(2, exponent) Range: [0, 31]",
              "direction": "in",
              "name": "exponent",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Rounded result of division. Midpoint is rounded away from zero."
            }
          ],
          "signature": "static int32_t arm_nn_divide_by_power_of_two(const int32_t dividend, const int32_t exponent)",
          "source": {
            "line": 2323,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2323"
          },
          "summary": "Rounding divide by power of two."
        },
        {
          "description": "Rounding divide by power of two for non-negative values.",
          "examples": [],
          "id": "arm_nn_nonneg_divide_by_pot_s32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_nonneg_divide_by_pot_s32",
          "params": [
            {
              "description": "- Dividend (assumed to be non-negative)",
              "direction": "in",
              "name": "dividend",
              "type": "int32_t"
            },
            {
              "description": "- Divisor = power(2, exponent) Range: [0, 31]",
              "direction": "in",
              "name": "exponent",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Rounded result of division. Midpoint is rounded away from zero."
            }
          ],
          "signature": "static int32_t arm_nn_nonneg_divide_by_pot_s32(int32_t dividend, int32_t exponent)",
          "source": {
            "line": 2378,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2378"
          },
          "summary": "Rounding divide by power of two for non-negative values."
        },
        {
          "description": "Requantize a given value.\n\nEssentially returns (val * multiplier)/(2 ^ shift) with different rounding depending if CMSIS_NN_USE_SINGLE_ROUNDING is defined or not.",
          "examples": [],
          "id": "arm_nn_requantize",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_requantize",
          "params": [
            {
              "description": "Value to be requantized",
              "direction": "in",
              "name": "val",
              "type": "const int32_t"
            },
            {
              "description": "Multiplier. Range {NN_Q31_MIN + 1, Q32_MAX}",
              "direction": "in",
              "name": "multiplier",
              "type": "const int32_t"
            },
            {
              "description": "Shift. Range: {-31, 30} Default branch: If shift is positive left shift 'val * multiplier' with shift If shift is negative right shift 'val * multiplier' with abs(shift) Single round branch: Input for total_shift in divide by '2 ^ total_shift'",
              "direction": "in",
              "name": "shift",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Default branch: Returns (val * multiplier) with rounding divided by (2 ^ shift) with rounding Single round branch: Returns (val * multiplier)/(2 ^ (31 - shift)) with rounding"
            }
          ],
          "signature": "static int32_t arm_nn_requantize(const int32_t val, const int32_t multiplier, const int32_t shift)",
          "source": {
            "line": 2416,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2416"
          },
          "summary": "Requantize a given value."
        },
        {
          "description": "Requantize a given 64 bit value.",
          "examples": [],
          "id": "arm_nn_requantize_s64",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_requantize_s64",
          "params": [
            {
              "description": "Value to be requantized in the range {-(1<<47)} to {(1<<47) - 1}",
              "direction": "in",
              "name": "val",
              "type": "const int64_t"
            },
            {
              "description": "Reduced multiplier in the range {NN_Q31_MIN + 1, Q32_MAX} to {Q16_MIN + 1, Q16_MAX}",
              "direction": "in",
              "name": "reduced_multiplier",
              "type": "const int32_t"
            },
            {
              "description": "Left or right shift for 'val * multiplier' in the range {-31} to {7}",
              "direction": "in",
              "name": "shift",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Returns (val * multiplier)/(2 ^ shift)"
            }
          ],
          "signature": "static int32_t arm_nn_requantize_s64(const int64_t val, const int32_t reduced_multiplier, const int32_t shift)",
          "source": {
            "line": 2453,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2453"
          },
          "summary": "Requantize a given 64 bit value."
        },
        {
          "description": "Saturating left shift for int16_t.",
          "examples": [],
          "id": "arm_nn_sat_lshift_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_sat_lshift_s16",
          "params": [
            {
              "description": "value to be shifted",
              "direction": "in",
              "name": "x",
              "type": "int16_t"
            },
            {
              "description": "Nonpositive values return x; positive values multiply by 2^shift with s16 saturation.",
              "direction": "in",
              "name": "shift",
              "type": "int"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "shifted value"
            }
          ],
          "signature": "static int16_t arm_nn_sat_lshift_s16(int16_t x, int shift)",
          "source": {
            "line": 2471,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2471"
          },
          "summary": "Saturating left shift for int16t."
        },
        {
          "description": "Saturating *Rounding* Doubling High Mul (s16).\n\nMatches NEON SQRDMULH s16",
          "examples": [],
          "id": "arm_nn_sqrdmulh_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_sqrdmulh_s16",
          "params": [
            {
              "description": "Multiplicand",
              "direction": "in",
              "name": "a",
              "type": "int16_t"
            },
            {
              "description": "Multiplier",
              "direction": "in",
              "name": "b",
              "type": "int16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Result of multiplication."
            }
          ],
          "signature": "static int16_t arm_nn_sqrdmulh_s16(int16_t a, int16_t b)",
          "source": {
            "line": 2490,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2490"
          },
          "summary": "Saturating Rounding Doubling High Mul (s16)."
        },
        {
          "description": "Saturating **Non-rounded** Doubling High Mul (s16).\n\nMatches NEON SQDMULH s16",
          "examples": [],
          "id": "arm_nn_sqdmulh_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_sqdmulh_s16",
          "params": [
            {
              "description": "Multiplicand",
              "direction": "in",
              "name": "a",
              "type": "int16_t"
            },
            {
              "description": "Multiplier",
              "direction": "in",
              "name": "b",
              "type": "int16_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Result of multiplication."
            }
          ],
          "signature": "static int16_t arm_nn_sqdmulh_s16(int16_t a, int16_t b)",
          "source": {
            "line": 2510,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2510"
          },
          "summary": "Saturating Non-rounded Doubling High Mul (s16)."
        },
        {
          "description": "Rounding divide by power of two (s16), midpoint away from zero.\n\nMirrors arm_nn_divide_by_power_of_two() semantics for s16.",
          "examples": [],
          "id": "arm_nn_divide_by_power_of_two_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_divide_by_power_of_two_s16",
          "params": [
            {
              "description": "Dividend",
              "direction": "in",
              "name": "x",
              "type": "int16_t"
            },
            {
              "description": "Divisor = power(2, exponent) Range: [0, 15]",
              "direction": "in",
              "name": "exponent",
              "type": "int"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Rounded result of division. Midpoint is rounded away from zero."
            }
          ],
          "signature": "static int16_t arm_nn_divide_by_power_of_two_s16(int16_t x, int exponent)",
          "source": {
            "line": 2530,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2530"
          },
          "summary": "Rounding divide by power of two (s16), midpoint away from zero."
        },
        {
          "description": "memcpy optimized for MVE",
          "examples": [],
          "id": "arm_memcpy_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memcpy_s8",
          "params": [
            {
              "description": "Destination pointer",
              "direction": "inout",
              "name": "dst",
              "type": "int8_t *"
            },
            {
              "description": "Source pointer.",
              "direction": "in",
              "name": "src",
              "type": "const int8_t *"
            },
            {
              "description": "Number of bytes to copy.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memcpy_s8(int8_t *dst, const int8_t *src, uint32_t block_size)",
          "source": {
            "line": 2545,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2545"
          },
          "summary": "memcpy optimized for MVE"
        },
        {
          "description": "memcpy optimized for MVE",
          "examples": [],
          "id": "arm_memcpy_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memcpy_s16",
          "params": [
            {
              "description": "Destination pointer",
              "direction": "inout",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "Source pointer.",
              "direction": "in",
              "name": "src",
              "type": "const int16_t *"
            },
            {
              "description": "Number of values to copy.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memcpy_s16(int16_t *dst, const int16_t *src, uint32_t block_size)",
          "source": {
            "line": 2569,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2569"
          },
          "summary": "memcpy optimized for MVE"
        },
        {
          "description": "memcpy optimized for MVE",
          "examples": [],
          "id": "arm_memcpy_s32",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memcpy_s32",
          "params": [
            {
              "description": "Destination pointer",
              "direction": "inout",
              "name": "dst",
              "type": "int32_t *"
            },
            {
              "description": "Source pointer.",
              "direction": "in",
              "name": "src",
              "type": "const int32_t *"
            },
            {
              "description": "Number of values to copy.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memcpy_s32(int32_t *dst, const int32_t *src, uint32_t block_size)",
          "source": {
            "line": 2581,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2581"
          },
          "summary": "memcpy optimized for MVE"
        },
        {
          "description": "memcpy wrapper for int16",
          "examples": [],
          "id": "arm_memcpy_q15",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_memcpy_q15",
          "params": [
            {
              "description": "Destination pointer",
              "direction": "inout",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "Source pointer.",
              "direction": "in",
              "name": "src",
              "type": "const int16_t *"
            },
            {
              "description": "Number of bytes to copy.",
              "direction": "in",
              "name": "block_size",
              "type": "uint32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_memcpy_q15(int16_t *dst, const int16_t *src, uint32_t block_size)",
          "source": {
            "line": 2593,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2593"
          },
          "summary": "memcpy wrapper for int16"
        },
        {
          "description": "Fixed-point exp() of a non-positive value.",
          "examples": [],
          "id": "arm_nn_exp_on_negative_values",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_exp_on_negative_values",
          "params": [
            {
              "description": "Input in Q5.26 fixed point. Must be less than or equal to 0",
              "direction": "in",
              "name": "val",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "exp(val) in Q0.31 fixed point. Returns NN_Q31_MAX when `val` is 0."
            }
          ],
          "signature": "static int32_t arm_nn_exp_on_negative_values(int32_t val)",
          "source": {
            "line": 2865,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2865"
          },
          "summary": "Fixed-point exp() of a non-positive value."
        },
        {
          "description": "",
          "examples": [],
          "id": "SELECT_IF_NON_ZERO",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "SELECT_IF_NON_ZERO",
          "params": [
            {
              "description": "",
              "name": "x"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "#define SELECT_IF_NON_ZERO(x) { \\ mask = MASK_IF_NON_ZERO(remainder & (1 << shift++)); \\ result = SELECT_USING_MASK(mask, MUL_SAT(result, x), result); \\ }",
          "source": {
            "line": 2878,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2878"
          },
          "summary": ""
        },
        {
          "description": "Saturating multiply by a power of two.",
          "examples": [],
          "id": "arm_nn_mult_by_power_of_two",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_mult_by_power_of_two",
          "params": [
            {
              "description": "Value to be multiplied",
              "direction": "in",
              "name": "val",
              "type": "const int32_t"
            },
            {
              "description": "Exponent. Multiplier = power(2, exp)",
              "direction": "in",
              "name": "exp",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "val * 2^exp saturated to the int32 range"
            }
          ],
          "signature": "static int32_t arm_nn_mult_by_power_of_two(const int32_t val, const int32_t exp)",
          "source": {
            "line": 2905,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2905"
          },
          "summary": "Saturating multiply by a power of two."
        },
        {
          "description": "Fixed-point 1 / (1 + x) for x in [0, 1), computed with Newton-Raphson iterations.",
          "examples": [],
          "id": "arm_nn_one_over_one_plus_x_for_x_in_0_1",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_one_over_one_plus_x_for_x_in_0_1",
          "params": [
            {
              "description": "x in Q0.31 fixed point. Range: [0, NN_Q31_MAX]",
              "direction": "in",
              "name": "val",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "1 / (1 + x) in Q0.31 fixed point"
            }
          ],
          "signature": "static int32_t arm_nn_one_over_one_plus_x_for_x_in_0_1(int32_t val)",
          "source": {
            "line": 2920,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2920"
          },
          "summary": "Fixed-point 1 / (1 + x) for x in [0, 1), computed with Newton-Raphson iterations."
        },
        {
          "description": "Write 2 s16 elements and post increment pointer.",
          "examples": [],
          "id": "arm_nn_write_q15x2_ia",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_write_q15x2_ia",
          "params": [
            {
              "description": "Pointer to pointer that holds address of destination. Advanced past the elements written.",
              "direction": "inout",
              "name": "dest_q15",
              "type": "int16_t **"
            },
            {
              "description": "Input value to be written.",
              "direction": "in",
              "name": "src_q31",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_nn_write_q15x2_ia(int16_t **dest_q15, int32_t src_q31)",
          "source": {
            "line": 2942,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2942"
          },
          "summary": "Write 2 s16 elements and post increment pointer."
        },
        {
          "description": "Write 2 s8 elements and post increment pointer.",
          "examples": [],
          "id": "arm_nn_write_s8x2_ia",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_write_s8x2_ia",
          "params": [
            {
              "description": "Pointer to pointer that holds address of destination. Advanced past the elements written.",
              "direction": "inout",
              "name": "dst",
              "type": "int8_t **"
            },
            {
              "description": "Input value to be written.",
              "direction": "in",
              "name": "src",
              "type": "int16_t"
            }
          ],
          "raises": [],
          "returns": [],
          "signature": "static void arm_nn_write_s8x2_ia(int8_t **dst, int16_t src)",
          "source": {
            "line": 2955,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2955"
          },
          "summary": "Write 2 s8 elements and post increment pointer."
        },
        {
          "description": "Get dimension value at specific index.",
          "examples": [],
          "id": "arm_cmsis_nn_dim_at",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_cmsis_nn_dim_at",
          "params": [
            {
              "description": "Pointer to `cmsis_nn_dims` structure",
              "direction": "in",
              "name": "dims",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "Index of dimension to get",
              "direction": "in",
              "name": "index",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Dimension value at specified index"
            }
          ],
          "signature": "static int32_t arm_cmsis_nn_dim_at(const cmsis_nn_dims *dims, int32_t index)",
          "source": {
            "line": 2969,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2969"
          },
          "summary": "Get dimension value at specific index."
        },
        {
          "description": "Calculate the product of all dimensions in a shape array.",
          "examples": [],
          "id": "arm_cmsis_nn_shape_product",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_cmsis_nn_shape_product",
          "params": [
            {
              "description": "Pointer to array containing shape dimensions",
              "direction": "in",
              "name": "shape",
              "type": "const int32_t *"
            },
            {
              "description": "Number of dimensions in the shape array",
              "direction": "in",
              "name": "length",
              "type": "int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "Product of all dimensions"
            }
          ],
          "signature": "static size_t arm_cmsis_nn_shape_product(const int32_t *shape, int32_t length)",
          "source": {
            "line": 2994,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L2994"
          },
          "summary": "Calculate the product of all dimensions in a shape array."
        },
        {
          "description": "Update LSTM function for an iteration step using s8 input and output, and s16 internally.",
          "examples": [],
          "id": "arm_nn_lstm_step_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_lstm_step_s8",
          "params": [
            {
              "description": "Data input pointer",
              "direction": "in",
              "name": "data_in",
              "type": "const int8_t *"
            },
            {
              "description": "Hidden state/ recurrent input pointer",
              "direction": "in",
              "name": "hidden_in",
              "type": "const int8_t *"
            },
            {
              "description": "Hidden state/ recurrent output pointer",
              "direction": "out",
              "name": "hidden_out",
              "type": "int8_t *"
            },
            {
              "description": "Struct containg all information about the lstm operator, see arm_nn_types.",
              "direction": "in",
              "name": "params",
              "type": "const cmsis_nn_lstm_params *"
            },
            {
              "description": "Struct containg pointers to all temporary scratch buffers needed for the lstm operator, see arm_nn_types.",
              "direction": "inout",
              "name": "buffers",
              "type": "cmsis_nn_lstm_context *"
            },
            {
              "description": "Number of timesteps between consecutive batches. E.g for params->timing_major = true, all batches for t=0 are stored sequentially, so batch offset = 1. For params->time major = false, all time steps are stored continously before the next batch, so batch offset = params->time_steps.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns ARM_CMSIS_NN_SUCCESS"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_lstm_step_s8(\n    const int8_t *data_in,\n    const int8_t *hidden_in,\n    int8_t *hidden_out,\n    const cmsis_nn_lstm_params *params,\n    cmsis_nn_lstm_context *buffers,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 3022,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3022"
          },
          "summary": "Update LSTM function for an iteration step using s8 input and output, and s16 internally."
        },
        {
          "description": "Update LSTM function for an iteration step using s16 input and output, and s16 internally.",
          "examples": [],
          "id": "arm_nn_lstm_step_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_lstm_step_s16",
          "params": [
            {
              "description": "Data input pointer",
              "direction": "in",
              "name": "data_in",
              "type": "const int16_t *"
            },
            {
              "description": "Hidden state/ recurrent input pointer",
              "direction": "in",
              "name": "hidden_in",
              "type": "const int16_t *"
            },
            {
              "description": "Hidden state/ recurrent output pointer",
              "direction": "out",
              "name": "hidden_out",
              "type": "int16_t *"
            },
            {
              "description": "Struct containg all information about the lstm operator, see arm_nn_types.",
              "direction": "in",
              "name": "params",
              "type": "const cmsis_nn_lstm_params *"
            },
            {
              "description": "Struct containg pointers to all temporary scratch buffers needed for the lstm operator, see arm_nn_types.",
              "direction": "inout",
              "name": "buffers",
              "type": "cmsis_nn_lstm_context *"
            },
            {
              "description": "Number of timesteps between consecutive batches. E.g for params->timing_major = true, all batches for t=0 are stored sequentially, so batch offset = 1. For params->time major = false, all time steps are stored continously before the next batch, so batch offset = params->time_steps.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns ARM_CMSIS_NN_SUCCESS"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_lstm_step_s16(\n    const int16_t *data_in,\n    const int16_t *hidden_in,\n    int16_t *hidden_out,\n    const cmsis_nn_lstm_params *params,\n    cmsis_nn_lstm_context *buffers,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 3046,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3046"
          },
          "summary": "Update LSTM function for an iteration step using s16 input and output, and s16 internally."
        },
        {
          "description": "Updates a LSTM gate for an iteration step of LSTM function, int8x8_16 version.",
          "examples": [],
          "id": "arm_nn_lstm_calculate_gate_s8_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_lstm_calculate_gate_s8_s16",
          "params": [
            {
              "description": "Data input pointer",
              "direction": "in",
              "name": "data_in",
              "type": "const int8_t *"
            },
            {
              "description": "Hidden state/ recurrent input pointer",
              "direction": "in",
              "name": "hidden_in",
              "type": "const int8_t *"
            },
            {
              "description": "Struct containing all information about the gate caluclation, see arm_nn_types.",
              "direction": "in",
              "name": "gate_data",
              "type": "const cmsis_nn_lstm_gate *"
            },
            {
              "description": "Struct containing all information about the lstm_operation, see arm_nn_types",
              "direction": "in",
              "name": "params",
              "type": "const cmsis_nn_lstm_params *"
            },
            {
              "description": "Hidden state/ recurrent output pointer",
              "direction": "out",
              "name": "output",
              "type": "int16_t *"
            },
            {
              "description": "Number of timesteps between consecutive batches, see arm_nn_lstm_step_s8.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns ARM_CMSIS_NN_SUCCESS"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_lstm_calculate_gate_s8_s16(\n    const int8_t *data_in,\n    const int8_t *hidden_in,\n    const cmsis_nn_lstm_gate *gate_data,\n    const cmsis_nn_lstm_params *params,\n    int16_t *output,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 3067,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3067"
          },
          "summary": "Updates a LSTM gate for an iteration step of LSTM function, int8x816 version."
        },
        {
          "description": "Updates a LSTM gate for an iteration step of LSTM function, int16x8_16 version.",
          "examples": [],
          "id": "arm_nn_lstm_calculate_gate_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_lstm_calculate_gate_s16",
          "params": [
            {
              "description": "Data input pointer",
              "direction": "in",
              "name": "data_in",
              "type": "const int16_t *"
            },
            {
              "description": "Hidden state/ recurrent input pointer",
              "direction": "in",
              "name": "hidden_in",
              "type": "const int16_t *"
            },
            {
              "description": "Struct containing all information about the gate caluclation, see arm_nn_types.",
              "direction": "in",
              "name": "gate_data",
              "type": "const cmsis_nn_lstm_gate *"
            },
            {
              "description": "Struct containing all information about the lstm_operation, see arm_nn_types",
              "direction": "in",
              "name": "params",
              "type": "const cmsis_nn_lstm_params *"
            },
            {
              "description": "Hidden state/ recurrent output pointer",
              "direction": "out",
              "name": "output",
              "type": "int16_t *"
            },
            {
              "description": "Number of timesteps between consecutive batches, see arm_nn_lstm_step_s16.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns ARM_CMSIS_NN_SUCCESS"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_lstm_calculate_gate_s16(\n    const int16_t *data_in,\n    const int16_t *hidden_in,\n    const cmsis_nn_lstm_gate *gate_data,\n    const cmsis_nn_lstm_params *params,\n    int16_t *output,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 3088,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3088"
          },
          "summary": "Updates a LSTM gate for an iteration step of LSTM function, int16x816 version."
        },
        {
          "description": "The result of the multiplication is accumulated to the passed result buffer. Multiplies a matrix by a \"batched\" vector (i.e. a matrix with a batch dimension composed by input vectors independent from each other).",
          "examples": [],
          "id": "arm_nn_vec_mat_mul_result_acc_s8_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_vec_mat_mul_result_acc_s8_s16",
          "params": [
            {
              "description": "Batched vector",
              "direction": "in",
              "name": "lhs",
              "type": "const int8_t *"
            },
            {
              "description": "Weights - input matrix (H(Rows)xW(Columns))",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Bias + lhs_offset * kernel_sum term precalculated into a constant vector.",
              "direction": "in",
              "name": "effective_bias",
              "type": "const int32_t *"
            },
            {
              "description": "Output",
              "direction": "out",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "Multiplier for quantization",
              "direction": "in",
              "name": "dst_multiplier",
              "type": "const int32_t"
            },
            {
              "description": "Shift for quantization",
              "direction": "in",
              "name": "dst_shift",
              "type": "const int32_t"
            },
            {
              "description": "Vector/matarix column length",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Row count of matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Batch size",
              "direction": "in",
              "name": "batches",
              "type": "const int32_t"
            },
            {
              "description": "Number of timesteps between consecutive batches in input, see arm_nn_lstm_step_s8. Note that the output is always stored with sequential batches.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_vec_mat_mul_result_acc_s8_s16(\n    const int8_t *lhs,\n    const int8_t *rhs,\n    const int32_t *effective_bias,\n    int16_t *dst,\n    const int32_t dst_multiplier,\n    const int32_t dst_shift,\n    const int32_t rhs_cols,\n    const int32_t rhs_rows,\n    const int32_t batches,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 3114,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3114"
          },
          "summary": "The result of the multiplication is accumulated to the passed result buffer."
        },
        {
          "description": "The result of the multiplication is accumulated to the passed result buffer. Multiplies a matrix by a \"batched\" vector (i.e. a matrix with a batch dimension composed by input vectors independent from each other).",
          "examples": [],
          "id": "arm_nn_vec_mat_mul_result_acc_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_vec_mat_mul_result_acc_s16",
          "params": [
            {
              "description": "Batched vector",
              "direction": "in",
              "name": "lhs",
              "type": "const int16_t *"
            },
            {
              "description": "Weights - input matrix (H(Rows)xW(Columns))",
              "direction": "in",
              "name": "rhs",
              "type": "const int8_t *"
            },
            {
              "description": "Bias + lhs_offset * kernel_sum term precalculated into a constant vector.",
              "direction": "in",
              "name": "effective_bias",
              "type": "const int64_t *"
            },
            {
              "description": "Output",
              "direction": "out",
              "name": "dst",
              "type": "int16_t *"
            },
            {
              "description": "Multiplier for quantization",
              "direction": "in",
              "name": "dst_multiplier",
              "type": "const int32_t"
            },
            {
              "description": "Shift for quantization",
              "direction": "in",
              "name": "dst_shift",
              "type": "const int32_t"
            },
            {
              "description": "Vector/matarix column length",
              "direction": "in",
              "name": "rhs_cols",
              "type": "const int32_t"
            },
            {
              "description": "Row count of matrix",
              "direction": "in",
              "name": "rhs_rows",
              "type": "const int32_t"
            },
            {
              "description": "Batch size",
              "direction": "in",
              "name": "batches",
              "type": "const int32_t"
            },
            {
              "description": "Number of timesteps between consecutive batches in input, see arm_nn_lstm_step_s16. Note that the output is always stored with sequential batches.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns `ARM_CMSIS_NN_SUCCESS`"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_nn_vec_mat_mul_result_acc_s16(\n    const int16_t *lhs,\n    const int8_t *rhs,\n    const int64_t *effective_bias,\n    int16_t *dst,\n    const int32_t dst_multiplier,\n    const int32_t dst_shift,\n    const int32_t rhs_cols,\n    const int32_t rhs_rows,\n    const int32_t batches,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 3144,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3144"
          },
          "summary": "The result of the multiplication is accumulated to the passed result buffer."
        },
        {
          "description": "s16 elementwise multiplication with s8 output\n\nSupported framework: TensorFlow Lite micro",
          "examples": [],
          "id": "arm_elementwise_mul_s16_s8",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_elementwise_mul_s16_s8",
          "params": [
            {
              "description": "pointer to input vector 1",
              "direction": "in",
              "name": "input_1_vect",
              "type": "const int16_t *"
            },
            {
              "description": "pointer to input vector 2",
              "direction": "in",
              "name": "input_2_vect",
              "type": "const int16_t *"
            },
            {
              "description": "pointer to output vector",
              "direction": "inout",
              "name": "output",
              "type": "int8_t *"
            },
            {
              "description": "output offset",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "output multiplier",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t"
            },
            {
              "description": "output shift",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t"
            },
            {
              "description": "number of samples per batch",
              "direction": "in",
              "name": "block_size",
              "type": "const int32_t"
            },
            {
              "description": "number of samples per batch",
              "direction": "in",
              "name": "batch_size",
              "type": "const int32_t"
            },
            {
              "description": "Number of timesteps between consecutive batches in output, see arm_nn_lstm_step_s8. Note that it is assumed that the input is stored with sequential batches.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns ARM_CMSIS_NN_SUCCESS"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_elementwise_mul_s16_s8(\n    const int16_t *input_1_vect,\n    const int16_t *input_2_vect,\n    int8_t *output,\n    const int32_t out_offset,\n    const int32_t out_mult,\n    const int32_t out_shift,\n    const int32_t block_size,\n    const int32_t batch_size,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 3171,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3171"
          },
          "summary": "s16 elementwise multiplication with s8 output"
        },
        {
          "description": "s16 elementwise multiplication with s16 output\n\nSupported framework: TensorFlow Lite micro",
          "examples": [],
          "id": "arm_elementwise_mul_s16_batch_offset",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_elementwise_mul_s16_batch_offset",
          "params": [
            {
              "description": "pointer to input vector 1",
              "direction": "in",
              "name": "input_1_vect",
              "type": "const int16_t *"
            },
            {
              "description": "pointer to input vector 2",
              "direction": "in",
              "name": "input_2_vect",
              "type": "const int16_t *"
            },
            {
              "description": "pointer to output vector",
              "direction": "inout",
              "name": "output",
              "type": "int16_t *"
            },
            {
              "description": "output offset",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "output multiplier",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t"
            },
            {
              "description": "output shift",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t"
            },
            {
              "description": "number of samples per batch",
              "direction": "in",
              "name": "block_size",
              "type": "const int32_t"
            },
            {
              "description": "number of samples per batch",
              "direction": "in",
              "name": "batch_size",
              "type": "const int32_t"
            },
            {
              "description": "Number of timesteps between consecutive batches in output, see arm_nn_lstm_step_s16. Note that it is assumed that the input is stored with sequential batches.",
              "direction": "in",
              "name": "batch_offset",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns ARM_CMSIS_NN_SUCCESS"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_elementwise_mul_s16_batch_offset(\n    const int16_t *input_1_vect,\n    const int16_t *input_2_vect,\n    int16_t *output,\n    const int32_t out_offset,\n    const int32_t out_mult,\n    const int32_t out_shift,\n    const int32_t block_size,\n    const int32_t batch_size,\n    const int32_t batch_offset\n)",
          "source": {
            "line": 3197,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3197"
          },
          "summary": "s16 elementwise multiplication with s16 output"
        },
        {
          "description": "s16 elementwise multiplication. The result of the multiplication is accumulated to the passed result buffer.\n\nSupported framework: TensorFlow Lite micro",
          "examples": [],
          "id": "arm_elementwise_mul_acc_s16",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_elementwise_mul_acc_s16",
          "params": [
            {
              "description": "pointer to input vector 1",
              "direction": "in",
              "name": "input_1_vect",
              "type": "const int16_t *"
            },
            {
              "description": "pointer to input vector 2",
              "direction": "in",
              "name": "input_2_vect",
              "type": "const int16_t *"
            },
            {
              "description": "offset for input 1. Not used.",
              "direction": "in",
              "name": "input_1_offset",
              "type": "const int32_t"
            },
            {
              "description": "offset for input 2. Not used.",
              "direction": "in",
              "name": "input_2_offset",
              "type": "const int32_t"
            },
            {
              "description": "pointer to output vector",
              "direction": "inout",
              "name": "output",
              "type": "int16_t *"
            },
            {
              "description": "output offset. Not used.",
              "direction": "in",
              "name": "out_offset",
              "type": "const int32_t"
            },
            {
              "description": "output multiplier",
              "direction": "in",
              "name": "out_mult",
              "type": "const int32_t"
            },
            {
              "description": "output shift",
              "direction": "in",
              "name": "out_shift",
              "type": "const int32_t"
            },
            {
              "description": "minimum value to clamp output to. Min: -32768",
              "direction": "in",
              "name": "out_activation_min",
              "type": "const int32_t"
            },
            {
              "description": "maximum value to clamp output to. Max: 32767",
              "direction": "in",
              "name": "out_activation_max",
              "type": "const int32_t"
            },
            {
              "description": "number of samples",
              "direction": "in",
              "name": "block_size",
              "type": "const int32_t"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns ARM_CMSIS_NN_SUCCESS"
            }
          ],
          "signature": "arm_cmsis_nn_status arm_elementwise_mul_acc_s16(\n    const int16_t *input_1_vect,\n    const int16_t *input_2_vect,\n    const int32_t input_1_offset,\n    const int32_t input_2_offset,\n    int16_t *output,\n    const int32_t out_offset,\n    const int32_t out_mult,\n    const int32_t out_shift,\n    const int32_t out_activation_min,\n    const int32_t out_activation_max,\n    const int32_t block_size\n)",
          "source": {
            "line": 3224,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3224"
          },
          "summary": "s16 elementwise multiplication."
        },
        {
          "description": "Check if a broadcast is required between 2 `cmsis_nn_dims`.\n\nCompares each dimension and returns 1 if any dimension does not match. This function does not check that broadcast rules are met.",
          "examples": [],
          "id": "arm_check_broadcast_required",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_check_broadcast_required",
          "params": [
            {
              "description": "pointer to input tensor 1",
              "direction": "in",
              "name": "shape_1",
              "type": "const cmsis_nn_dims *"
            },
            {
              "description": "pointer to input tensor 2",
              "direction": "in",
              "name": "shape_2",
              "type": "const cmsis_nn_dims *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "The function returns 1 if a broadcast is required, or 0 if not."
            }
          ],
          "signature": "static int32_t arm_check_broadcast_required(const cmsis_nn_dims *shape_1, const cmsis_nn_dims *shape_2)",
          "source": {
            "line": 3245,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3245"
          },
          "summary": "Check if a broadcast is required between 2 cmsisnndims."
        },
        {
          "description": "Reports whether the reduced axes of a 4-D tensor form one contiguous block followed by kept axes, as in a NHWC mean over H and W, and gives the flattened sizes. Axes of size 1 are ignored.",
          "examples": [],
          "id": "arm_reduce_get_middle_block_from_arrays",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_reduce_get_middle_block_from_arrays",
          "params": [
            {
              "description": "4-element array {n, h, w, c}",
              "direction": "in",
              "name": "in_dims",
              "type": "const int32_t"
            },
            {
              "description": "4-element mask {axis_n, axis_h, axis_w, axis_c}",
              "direction": "in",
              "name": "axis_arr",
              "type": "const int32_t"
            },
            {
              "description": "Product of the dims before the reduced block",
              "direction": "out",
              "name": "outer",
              "type": "int32_t *"
            },
            {
              "description": "Product of the reduced dims",
              "direction": "out",
              "name": "reduce",
              "type": "int32_t *"
            },
            {
              "description": "Product of the dims after the reduced block",
              "direction": "out",
              "name": "inner",
              "type": "int32_t *"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "1 if the input is [outer, reduce, inner] with the middle dim reduced and inner > 1, otherwise 0"
            }
          ],
          "signature": "static int32_t arm_reduce_get_middle_block_from_arrays(\n    const int32_t in_dims,\n    const int32_t axis_arr,\n    int32_t *outer,\n    int32_t *reduce,\n    int32_t *inner\n)",
          "source": {
            "line": 3310,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3310"
          },
          "summary": "Reports whether the reduced axes of a 4-D tensor form one contiguous block followed by kept axes, as in a NHWC mean over H and W, and gives the flattened sizes."
        },
        {
          "description": "",
          "examples": [],
          "id": "ARM_NN_SQRT_S16_TABLEFREE_SHIFT",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_SQRT_S16_TABLEFREE_SHIFT",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_SQRT_S16_TABLEFREE_SHIFT 14",
          "source": {
            "line": 3364,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3364"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "ARM_NN_SQRT_S16_TABLEFREE_MAGIC",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_SQRT_S16_TABLEFREE_MAGIC",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_SQRT_S16_TABLEFREE_MAGIC UINT32_C(0x5F5FB6C4)",
          "source": {
            "line": 3365,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3365"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "ARM_NN_SQRT_S16_TABLEFREE_K0",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_SQRT_S16_TABLEFREE_K0",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_SQRT_S16_TABLEFREE_K0 (-4.76426697f)",
          "source": {
            "line": 3366,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3366"
          },
          "summary": ""
        },
        {
          "description": "",
          "examples": [],
          "id": "ARM_NN_SQRT_S16_TABLEFREE_K1",
          "kind": "macro",
          "language": "c",
          "members": [],
          "name": "ARM_NN_SQRT_S16_TABLEFREE_K1",
          "params": [],
          "raises": [],
          "returns": [],
          "signature": "#define ARM_NN_SQRT_S16_TABLEFREE_K1 (-48.0000114f)",
          "source": {
            "line": 3367,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3367"
          },
          "summary": ""
        },
        {
          "description": "One element of `arm_sqrt_s16_tablefree()`: the float32 chain the MVE path evaluates per lane, so the two agree bit for bit on any IEEE-754 float32 implementation with round-to-nearest-even and a fused multiply-add (fmaf). Every product after the pre-scale either has two uses or feeds an fmaf or a conversion, never another lone multiply, so a compiler allowed to reassociate (-ffast-math) still has no chain to reorder, and no product feeds a bare add, so there is nothing to contract.",
          "examples": [],
          "id": "arm_nn_sqrt_s16_tablefree_element",
          "kind": "function",
          "language": "c",
          "members": [],
          "name": "arm_nn_sqrt_s16_tablefree_element",
          "params": [
            {
              "description": "input code; values <= 0 give 0",
              "direction": "in",
              "name": "value",
              "type": "const int32_t"
            },
            {
              "description": "input_scale / (output_scale * output_scale) as float32",
              "direction": "in",
              "name": "scale",
              "type": "const float"
            }
          ],
          "raises": [],
          "returns": [
            {
              "description": "trunc(sqrt(value * scale)) saturated to 32767"
            }
          ],
          "signature": "static int16_t arm_nn_sqrt_s16_tablefree_element(const int32_t value, const float scale)",
          "source": {
            "line": 3381,
            "path": "Include/arm_nnsupportfunctions.h",
            "url": "https://github.com/AmbiqAI/ns-cmsis-nn/blob/5f3fed9f21a57390cc7f00f77a37db8f5f110cb8/Include/arm_nnsupportfunctions.h#L3381"
          },
          "summary": "One element of armsqrts16tablefree(): the float32 chain the MVE path evaluates per lane, so the two agree bit for bit on any IEEE-754 float32 implementation wi…"
        }
      ]
    }
  ],
  "name": "heliaCORE"
}
