{"api_version":"1","snapshot":"2026-10-08","licence":"CC0-1.0","architecture":"AMDGPU","records":2442,"mnemonics":[{"mnemonic":"buffer_atomic_add","slug":"buffer_atomic_add","records":1,"summary":"Add two unsigned 32-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_add/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_add.json","aliases":["buffer_atomic_add_u32"]},{"mnemonic":"buffer_atomic_add_f32","slug":"buffer_atomic_add_f32","records":1,"summary":"Add two single-precision float values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_add_f32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_add_f32.json"},{"mnemonic":"buffer_atomic_add_f64","slug":"buffer_atomic_add_f64","records":1,"summary":"Add a double-precision float value in the data register to a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_add_f64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_add_f64.json"},{"mnemonic":"buffer_atomic_add_u32","slug":"buffer_atomic_add_u32","records":1,"summary":"Add two unsigned 32-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_add_u32.json","aliases":["buffer_atomic_add"]},{"mnemonic":"buffer_atomic_add_u64","slug":"buffer_atomic_add_u64","records":1,"summary":"Add two unsigned 64-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_add_u64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_add_u64.json","aliases":["buffer_atomic_add_x2"]},{"mnemonic":"buffer_atomic_add_x2","slug":"buffer_atomic_add_x2","records":1,"summary":"Add two unsigned 64-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_add_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_add_x2.json","aliases":["buffer_atomic_add_u64"]},{"mnemonic":"buffer_atomic_and","slug":"buffer_atomic_and","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_and/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_and.json","aliases":["buffer_atomic_and_b32"]},{"mnemonic":"buffer_atomic_and_b32","slug":"buffer_atomic_and_b32","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_and_b32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_and_b32.json","aliases":["buffer_atomic_and"]},{"mnemonic":"buffer_atomic_and_b64","slug":"buffer_atomic_and_b64","records":1,"summary":"Calculate bitwise AND given two unsigned 64-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_and_b64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_and_b64.json","aliases":["buffer_atomic_and_x2"]},{"mnemonic":"buffer_atomic_and_x2","slug":"buffer_atomic_and_x2","records":1,"summary":"Calculate bitwise AND given two unsigned 64-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_and_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_and_x2.json","aliases":["buffer_atomic_and_b64"]},{"mnemonic":"buffer_atomic_cmpswap","slug":"buffer_atomic_cmpswap","records":1,"summary":"Compare two unsigned 32-bit integer values stored in the data comparison register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_cmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_cmpswap.json","aliases":["buffer_atomic_cmpswap_b32"]},{"mnemonic":"buffer_atomic_cmpswap_b32","slug":"buffer_atomic_cmpswap_b32","records":1,"summary":"Compare two unsigned 32-bit integer values stored in the data comparison register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_cmpswap_b32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_cmpswap_b32.json","aliases":["buffer_atomic_cmpswap"]},{"mnemonic":"buffer_atomic_cmpswap_b64","slug":"buffer_atomic_cmpswap_b64","records":1,"summary":"Compare two unsigned 64-bit integer values stored in the data comparison register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_cmpswap_b64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_cmpswap_b64.json","aliases":["buffer_atomic_cmpswap_x2"]},{"mnemonic":"buffer_atomic_cmpswap_f32","slug":"buffer_atomic_cmpswap_f32","records":1,"summary":"Compare two single-precision float values stored in the data comparison register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_cmpswap_f32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_cmpswap_f32.json","aliases":["buffer_atomic_fcmpswap"]},{"mnemonic":"buffer_atomic_cmpswap_x2","slug":"buffer_atomic_cmpswap_x2","records":1,"summary":"Compare two unsigned 64-bit integer values stored in the data comparison register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_cmpswap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_cmpswap_x2.json","aliases":["buffer_atomic_cmpswap_b64"]},{"mnemonic":"buffer_atomic_cond_sub_u32","slug":"buffer_atomic_cond_sub_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value in the data register from a location in a buffer surface only if the memory value is greater than or equal…","page":"https://instructionsets.com/amdgpu/buffer_atomic_cond_sub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_cond_sub_u32.json"},{"mnemonic":"buffer_atomic_csub","slug":"buffer_atomic_csub","records":1,"summary":"Subtract an unsigned 32-bit integer location in a buffer surface from a value in the data register and clamp the result to zero.","page":"https://instructionsets.com/amdgpu/buffer_atomic_csub/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_csub.json","aliases":["buffer_atomic_csub_u32","buffer_atomic_sub_clamp_u32"]},{"mnemonic":"buffer_atomic_csub_u32","slug":"buffer_atomic_csub_u32","records":1,"summary":"Subtract an unsigned 32-bit integer location in a buffer surface from a value in the data register and clamp the result to zero.","page":"https://instructionsets.com/amdgpu/buffer_atomic_csub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_csub_u32.json","aliases":["buffer_atomic_csub","buffer_atomic_sub_clamp_u32"]},{"mnemonic":"buffer_atomic_dec","slug":"buffer_atomic_dec","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in a buffer surface with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/buffer_atomic_dec/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_dec.json","aliases":["buffer_atomic_dec_u32"]},{"mnemonic":"buffer_atomic_dec_u32","slug":"buffer_atomic_dec_u32","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in a buffer surface with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/buffer_atomic_dec_u32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_dec_u32.json","aliases":["buffer_atomic_dec"]},{"mnemonic":"buffer_atomic_dec_u64","slug":"buffer_atomic_dec_u64","records":1,"summary":"Decrement an unsigned 64-bit integer value from a location in a buffer surface with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/buffer_atomic_dec_u64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_dec_u64.json","aliases":["buffer_atomic_dec_x2"]},{"mnemonic":"buffer_atomic_dec_x2","slug":"buffer_atomic_dec_x2","records":1,"summary":"Decrement an unsigned 64-bit integer value from a location in a buffer surface with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/buffer_atomic_dec_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_dec_x2.json","aliases":["buffer_atomic_dec_u64"]},{"mnemonic":"buffer_atomic_fcmpswap","slug":"buffer_atomic_fcmpswap","records":1,"summary":"Compare two single-precision float values stored in the data comparison register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_fcmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_fcmpswap.json","aliases":["buffer_atomic_cmpswap_f32"]},{"mnemonic":"buffer_atomic_fcmpswap_x2","slug":"buffer_atomic_fcmpswap_x2","records":1,"summary":"Compare two double-precision float values stored in the data comparison register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_fcmpswap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_fcmpswap_x2.json"},{"mnemonic":"buffer_atomic_fmax","slug":"buffer_atomic_fmax","records":1,"summary":"Select the maximum of two single-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_fmax/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_fmax.json","aliases":["buffer_atomic_max_f32","buffer_atomic_max_num_f32"]},{"mnemonic":"buffer_atomic_fmax_x2","slug":"buffer_atomic_fmax_x2","records":1,"summary":"Select the maximum of two double-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_fmax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_fmax_x2.json"},{"mnemonic":"buffer_atomic_fmin","slug":"buffer_atomic_fmin","records":1,"summary":"Select the minimum of two single-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_fmin/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_fmin.json","aliases":["buffer_atomic_min_f32","buffer_atomic_min_num_f32"]},{"mnemonic":"buffer_atomic_fmin_x2","slug":"buffer_atomic_fmin_x2","records":1,"summary":"Select the minimum of two double-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_fmin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_fmin_x2.json"},{"mnemonic":"buffer_atomic_inc","slug":"buffer_atomic_inc","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in a buffer surface with wraparound to 0 if the value exceeds a value in the data register.","page":"https://instructionsets.com/amdgpu/buffer_atomic_inc/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_inc.json","aliases":["buffer_atomic_inc_u32"]},{"mnemonic":"buffer_atomic_inc_u32","slug":"buffer_atomic_inc_u32","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in a buffer surface with wraparound to 0 if the value exceeds a value in the data register.","page":"https://instructionsets.com/amdgpu/buffer_atomic_inc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_inc_u32.json","aliases":["buffer_atomic_inc"]},{"mnemonic":"buffer_atomic_inc_u64","slug":"buffer_atomic_inc_u64","records":1,"summary":"Increment an unsigned 64-bit integer value from a location in a buffer surface with wraparound to 0 if the value exceeds a value in the data register.","page":"https://instructionsets.com/amdgpu/buffer_atomic_inc_u64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_inc_u64.json","aliases":["buffer_atomic_inc_x2"]},{"mnemonic":"buffer_atomic_inc_x2","slug":"buffer_atomic_inc_x2","records":1,"summary":"Increment an unsigned 64-bit integer value from a location in a buffer surface with wraparound to 0 if the value exceeds a value in the data register.","page":"https://instructionsets.com/amdgpu/buffer_atomic_inc_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_inc_x2.json","aliases":["buffer_atomic_inc_u64"]},{"mnemonic":"buffer_atomic_max_f32","slug":"buffer_atomic_max_f32","records":1,"summary":"Select the maximum of two single-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_max_f32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_max_f32.json","aliases":["buffer_atomic_fmax","buffer_atomic_max_num_f32"]},{"mnemonic":"buffer_atomic_max_f64","slug":"buffer_atomic_max_f64","records":1,"summary":"Select the maximum of two double-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_max_f64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_max_f64.json"},{"mnemonic":"buffer_atomic_max_i32","slug":"buffer_atomic_max_i32","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_max_i32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_max_i32.json","aliases":["buffer_atomic_smax"]},{"mnemonic":"buffer_atomic_max_i64","slug":"buffer_atomic_max_i64","records":1,"summary":"Select the maximum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_max_i64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_max_i64.json","aliases":["buffer_atomic_smax_x2"]},{"mnemonic":"buffer_atomic_max_num_f32","slug":"buffer_atomic_max_num_f32","records":1,"summary":"Select the IEEE maximumNumber() of two single-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_max_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_max_num_f32.json","aliases":["buffer_atomic_fmax","buffer_atomic_max_f32"]},{"mnemonic":"buffer_atomic_max_num_f64","slug":"buffer_atomic_max_num_f64","records":1,"summary":"AMDGPU MUBUF vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/buffer_atomic_max_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_max_num_f64.json"},{"mnemonic":"buffer_atomic_max_u32","slug":"buffer_atomic_max_u32","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_max_u32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_max_u32.json","aliases":["buffer_atomic_umax"]},{"mnemonic":"buffer_atomic_max_u64","slug":"buffer_atomic_max_u64","records":1,"summary":"Select the maximum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_max_u64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_max_u64.json","aliases":["buffer_atomic_umax_x2"]},{"mnemonic":"buffer_atomic_min_f32","slug":"buffer_atomic_min_f32","records":1,"summary":"Select the minimum of two single-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_min_f32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_min_f32.json","aliases":["buffer_atomic_fmin","buffer_atomic_min_num_f32"]},{"mnemonic":"buffer_atomic_min_f64","slug":"buffer_atomic_min_f64","records":1,"summary":"Select the minimum of two double-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_min_f64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_min_f64.json"},{"mnemonic":"buffer_atomic_min_i32","slug":"buffer_atomic_min_i32","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_min_i32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_min_i32.json","aliases":["buffer_atomic_smin"]},{"mnemonic":"buffer_atomic_min_i64","slug":"buffer_atomic_min_i64","records":1,"summary":"Select the minimum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_min_i64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_min_i64.json","aliases":["buffer_atomic_smin_x2"]},{"mnemonic":"buffer_atomic_min_num_f32","slug":"buffer_atomic_min_num_f32","records":1,"summary":"Select the IEEE minimumNumber() of two single-precision float inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_min_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_min_num_f32.json","aliases":["buffer_atomic_fmin","buffer_atomic_min_f32"]},{"mnemonic":"buffer_atomic_min_num_f64","slug":"buffer_atomic_min_num_f64","records":1,"summary":"AMDGPU MUBUF vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/buffer_atomic_min_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_min_num_f64.json"},{"mnemonic":"buffer_atomic_min_u32","slug":"buffer_atomic_min_u32","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_min_u32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_min_u32.json","aliases":["buffer_atomic_umin"]},{"mnemonic":"buffer_atomic_min_u64","slug":"buffer_atomic_min_u64","records":1,"summary":"Select the minimum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_min_u64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_min_u64.json","aliases":["buffer_atomic_umin_x2"]},{"mnemonic":"buffer_atomic_or","slug":"buffer_atomic_or","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_or/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_or.json","aliases":["buffer_atomic_or_b32"]},{"mnemonic":"buffer_atomic_or_b32","slug":"buffer_atomic_or_b32","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_or_b32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_or_b32.json","aliases":["buffer_atomic_or"]},{"mnemonic":"buffer_atomic_or_b64","slug":"buffer_atomic_or_b64","records":1,"summary":"Calculate bitwise OR given two unsigned 64-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_or_b64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_or_b64.json","aliases":["buffer_atomic_or_x2"]},{"mnemonic":"buffer_atomic_or_x2","slug":"buffer_atomic_or_x2","records":1,"summary":"Calculate bitwise OR given two unsigned 64-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_or_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_or_x2.json","aliases":["buffer_atomic_or_b64"]},{"mnemonic":"buffer_atomic_pk_add_bf16","slug":"buffer_atomic_pk_add_bf16","records":1,"summary":"Add a packed 2-component BF16 float value from the data register to a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_pk_add_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_pk_add_bf16.json"},{"mnemonic":"buffer_atomic_pk_add_f16","slug":"buffer_atomic_pk_add_f16","records":1,"summary":"Add a packed 2-component half-precision float value from the data register to a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_pk_add_f16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_pk_add_f16.json"},{"mnemonic":"buffer_atomic_rsub","slug":"buffer_atomic_rsub","records":1,"summary":"AMDGPU MUBUF vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/buffer_atomic_rsub/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_rsub.json"},{"mnemonic":"buffer_atomic_rsub_x2","slug":"buffer_atomic_rsub_x2","records":1,"summary":"AMDGPU MUBUF vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/buffer_atomic_rsub_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_rsub_x2.json"},{"mnemonic":"buffer_atomic_smax","slug":"buffer_atomic_smax","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_smax/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_smax.json","aliases":["buffer_atomic_max_i32"]},{"mnemonic":"buffer_atomic_smax_x2","slug":"buffer_atomic_smax_x2","records":1,"summary":"Select the maximum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_smax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_smax_x2.json","aliases":["buffer_atomic_max_i64"]},{"mnemonic":"buffer_atomic_smin","slug":"buffer_atomic_smin","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_smin/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_smin.json","aliases":["buffer_atomic_min_i32"]},{"mnemonic":"buffer_atomic_smin_x2","slug":"buffer_atomic_smin_x2","records":1,"summary":"Select the minimum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_smin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_smin_x2.json","aliases":["buffer_atomic_min_i64"]},{"mnemonic":"buffer_atomic_sub","slug":"buffer_atomic_sub","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_sub/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_sub.json","aliases":["buffer_atomic_sub_u32"]},{"mnemonic":"buffer_atomic_sub_clamp_u32","slug":"buffer_atomic_sub_clamp_u32","records":1,"summary":"Subtract an unsigned 32-bit integer location in a buffer surface from a value in the data register and clamp the result to zero.","page":"https://instructionsets.com/amdgpu/buffer_atomic_sub_clamp_u32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_sub_clamp_u32.json","aliases":["buffer_atomic_csub","buffer_atomic_csub_u32"]},{"mnemonic":"buffer_atomic_sub_u32","slug":"buffer_atomic_sub_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_sub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_sub_u32.json","aliases":["buffer_atomic_sub"]},{"mnemonic":"buffer_atomic_sub_u64","slug":"buffer_atomic_sub_u64","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in the data register from a value stored in a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_sub_u64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_sub_u64.json","aliases":["buffer_atomic_sub_x2"]},{"mnemonic":"buffer_atomic_sub_x2","slug":"buffer_atomic_sub_x2","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in the data register from a value stored in a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_sub_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_sub_x2.json","aliases":["buffer_atomic_sub_u64"]},{"mnemonic":"buffer_atomic_swap","slug":"buffer_atomic_swap","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_swap/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_swap.json","aliases":["buffer_atomic_swap_b32"]},{"mnemonic":"buffer_atomic_swap_b32","slug":"buffer_atomic_swap_b32","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_swap_b32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_swap_b32.json","aliases":["buffer_atomic_swap"]},{"mnemonic":"buffer_atomic_swap_b64","slug":"buffer_atomic_swap_b64","records":1,"summary":"Swap an unsigned 64-bit integer value in the data register with a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_swap_b64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_swap_b64.json","aliases":["buffer_atomic_swap_x2"]},{"mnemonic":"buffer_atomic_swap_x2","slug":"buffer_atomic_swap_x2","records":1,"summary":"Swap an unsigned 64-bit integer value in the data register with a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_swap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_swap_x2.json","aliases":["buffer_atomic_swap_b64"]},{"mnemonic":"buffer_atomic_umax","slug":"buffer_atomic_umax","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_umax/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_umax.json","aliases":["buffer_atomic_max_u32"]},{"mnemonic":"buffer_atomic_umax_x2","slug":"buffer_atomic_umax_x2","records":1,"summary":"Select the maximum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_umax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_umax_x2.json","aliases":["buffer_atomic_max_u64"]},{"mnemonic":"buffer_atomic_umin","slug":"buffer_atomic_umin","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_umin/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_umin.json","aliases":["buffer_atomic_min_u32"]},{"mnemonic":"buffer_atomic_umin_x2","slug":"buffer_atomic_umin_x2","records":1,"summary":"Select the minimum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_umin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_umin_x2.json","aliases":["buffer_atomic_min_u64"]},{"mnemonic":"buffer_atomic_xor","slug":"buffer_atomic_xor","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_xor/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_xor.json","aliases":["buffer_atomic_xor_b32"]},{"mnemonic":"buffer_atomic_xor_b32","slug":"buffer_atomic_xor_b32","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_xor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_xor_b32.json","aliases":["buffer_atomic_xor"]},{"mnemonic":"buffer_atomic_xor_b64","slug":"buffer_atomic_xor_b64","records":1,"summary":"Calculate bitwise XOR given two unsigned 64-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_xor_b64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_xor_b64.json","aliases":["buffer_atomic_xor_x2"]},{"mnemonic":"buffer_atomic_xor_x2","slug":"buffer_atomic_xor_x2","records":1,"summary":"Calculate bitwise XOR given two unsigned 64-bit integer values stored in the data register and a location in a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_atomic_xor_x2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_atomic_xor_x2.json","aliases":["buffer_atomic_xor_b64"]},{"mnemonic":"buffer_gl0_inv","slug":"buffer_gl0_inv","records":1,"summary":"Write back and invalidate the shader L0. Returns ACK to shader.","page":"https://instructionsets.com/amdgpu/buffer_gl0_inv/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_gl0_inv.json"},{"mnemonic":"buffer_gl1_inv","slug":"buffer_gl1_inv","records":1,"summary":"Invalidate the GL1 cache only. Returns ACK to shader.","page":"https://instructionsets.com/amdgpu/buffer_gl1_inv/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_gl1_inv.json"},{"mnemonic":"buffer_inv","slug":"buffer_inv","records":1,"summary":"Invalidate CU and/or L2 cache depending on sc0 and sc1 bits. Returns ACK to shader.","page":"https://instructionsets.com/amdgpu/buffer_inv/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_inv.json","aliases":["buffer_invl2"]},{"mnemonic":"buffer_invl2","slug":"buffer_invl2","records":1,"summary":"Invalidate L2 cache. Returns ACK to shader.","page":"https://instructionsets.com/amdgpu/buffer_invl2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_invl2.json","aliases":["buffer_inv"]},{"mnemonic":"buffer_load_b128","slug":"buffer_load_b128","records":1,"summary":"Load 128 bits of data from a buffer surface into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_b128/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_b128.json","aliases":["buffer_load_dwordx4"]},{"mnemonic":"buffer_load_b32","slug":"buffer_load_b32","records":1,"summary":"Load 32 bits of data from a buffer surface into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_b32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_b32.json","aliases":["buffer_load_dword"]},{"mnemonic":"buffer_load_b64","slug":"buffer_load_b64","records":1,"summary":"Load 64 bits of data from a buffer surface into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_b64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_b64.json","aliases":["buffer_load_dwordx2"]},{"mnemonic":"buffer_load_b96","slug":"buffer_load_b96","records":1,"summary":"Load 96 bits of data from a buffer surface into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_b96/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_b96.json","aliases":["buffer_load_dwordx3"]},{"mnemonic":"buffer_load_d16_b16","slug":"buffer_load_d16_b16","records":1,"summary":"Load 16 bits of unsigned data from a buffer surface and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_d16_b16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_b16.json","aliases":["buffer_load_short_d16"]},{"mnemonic":"buffer_load_d16_format_x","slug":"buffer_load_d16_format_x","records":1,"summary":"Load 1-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_d16_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_format_x.json","aliases":["buffer_load_format_d16_x"]},{"mnemonic":"buffer_load_d16_format_xy","slug":"buffer_load_d16_format_xy","records":1,"summary":"Load 2-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_d16_format_xy/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_format_xy.json","aliases":["buffer_load_format_d16_xy"]},{"mnemonic":"buffer_load_d16_format_xyz","slug":"buffer_load_d16_format_xyz","records":1,"summary":"Load 3-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_d16_format_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_format_xyz.json","aliases":["buffer_load_format_d16_xyz"]},{"mnemonic":"buffer_load_d16_format_xyzw","slug":"buffer_load_d16_format_xyzw","records":1,"summary":"Load 4-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_d16_format_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_format_xyzw.json","aliases":["buffer_load_format_d16_xyzw"]},{"mnemonic":"buffer_load_d16_hi_b16","slug":"buffer_load_d16_hi_b16","records":1,"summary":"Load 16 bits of unsigned data from a buffer surface and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_d16_hi_b16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_hi_b16.json","aliases":["buffer_load_short_d16_hi"]},{"mnemonic":"buffer_load_d16_hi_format_x","slug":"buffer_load_d16_hi_format_x","records":1,"summary":"Load 1-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_d16_hi_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_hi_format_x.json","aliases":["buffer_load_format_d16_hi_x"]},{"mnemonic":"buffer_load_d16_hi_i8","slug":"buffer_load_d16_hi_i8","records":1,"summary":"Load 8 bits of signed data from a buffer surface, sign extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_d16_hi_i8/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_hi_i8.json","aliases":["buffer_load_sbyte_d16_hi"]},{"mnemonic":"buffer_load_d16_hi_u8","slug":"buffer_load_d16_hi_u8","records":1,"summary":"Load 8 bits of unsigned data from a buffer surface, zero extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_d16_hi_u8/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_hi_u8.json","aliases":["buffer_load_ubyte_d16_hi"]},{"mnemonic":"buffer_load_d16_i8","slug":"buffer_load_d16_i8","records":1,"summary":"Load 8 bits of signed data from a buffer surface, sign extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_d16_i8/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_i8.json","aliases":["buffer_load_sbyte_d16"]},{"mnemonic":"buffer_load_d16_u8","slug":"buffer_load_d16_u8","records":1,"summary":"Load 8 bits of unsigned data from a buffer surface, zero extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_d16_u8/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_d16_u8.json","aliases":["buffer_load_ubyte_d16"]},{"mnemonic":"buffer_load_dword","slug":"buffer_load_dword","records":1,"summary":"Load one 32-bit dword per lane through a buffer (raw/structured) resource descriptor.","page":"https://instructionsets.com/amdgpu/buffer_load_dword/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_dword.json","aliases":["buffer_load_b32"]},{"mnemonic":"buffer_load_dwordx2","slug":"buffer_load_dwordx2","records":1,"summary":"Load 64 bits of data from a buffer surface into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_dwordx2.json","aliases":["buffer_load_b64"]},{"mnemonic":"buffer_load_dwordx3","slug":"buffer_load_dwordx3","records":1,"summary":"Load 96 bits of data from a buffer surface into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_dwordx3/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_dwordx3.json","aliases":["buffer_load_b96"]},{"mnemonic":"buffer_load_dwordx4","slug":"buffer_load_dwordx4","records":1,"summary":"Load 128 bits of data from a buffer surface into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_dwordx4.json","aliases":["buffer_load_b128"]},{"mnemonic":"buffer_load_format_d16_hi_x","slug":"buffer_load_format_d16_hi_x","records":1,"summary":"Load 1-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_format_d16_hi_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_format_d16_hi_x.json","aliases":["buffer_load_d16_hi_format_x"]},{"mnemonic":"buffer_load_format_d16_x","slug":"buffer_load_format_d16_x","records":1,"summary":"Load 1-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_format_d16_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_format_d16_x.json","aliases":["buffer_load_d16_format_x"]},{"mnemonic":"buffer_load_format_d16_xy","slug":"buffer_load_format_d16_xy","records":1,"summary":"Load 2-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_format_d16_xy/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_format_d16_xy.json","aliases":["buffer_load_d16_format_xy"]},{"mnemonic":"buffer_load_format_d16_xyz","slug":"buffer_load_format_d16_xyz","records":1,"summary":"Load 3-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_format_d16_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_format_d16_xyz.json","aliases":["buffer_load_d16_format_xyz"]},{"mnemonic":"buffer_load_format_d16_xyzw","slug":"buffer_load_format_d16_xyzw","records":1,"summary":"Load 4-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/buffer_load_format_d16_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_format_d16_xyzw.json","aliases":["buffer_load_d16_format_xyzw"]},{"mnemonic":"buffer_load_format_x","slug":"buffer_load_format_x","records":1,"summary":"Load 1-component formatted data from a buffer surface, convert the data to 32 bit integral or floating point format, then store the result into a…","page":"https://instructionsets.com/amdgpu/buffer_load_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_format_x.json"},{"mnemonic":"buffer_load_format_xy","slug":"buffer_load_format_xy","records":1,"summary":"Load 2-component formatted data from a buffer surface, convert the data to 32 bit integral or floating point format, then store the result into a…","page":"https://instructionsets.com/amdgpu/buffer_load_format_xy/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_format_xy.json"},{"mnemonic":"buffer_load_format_xyz","slug":"buffer_load_format_xyz","records":1,"summary":"Load 3-component formatted data from a buffer surface, convert the data to 32 bit integral or floating point format, then store the result into a…","page":"https://instructionsets.com/amdgpu/buffer_load_format_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_format_xyz.json"},{"mnemonic":"buffer_load_format_xyzw","slug":"buffer_load_format_xyzw","records":1,"summary":"Load 4-component formatted data from a buffer surface, convert the data to 32 bit integral or floating point format, then store the result into a…","page":"https://instructionsets.com/amdgpu/buffer_load_format_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_format_xyzw.json"},{"mnemonic":"buffer_load_i16","slug":"buffer_load_i16","records":1,"summary":"Load 16 bits of signed data from a buffer surface, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_i16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_i16.json","aliases":["buffer_load_sshort"]},{"mnemonic":"buffer_load_i8","slug":"buffer_load_i8","records":1,"summary":"Load 8 bits of signed data from a buffer surface, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_i8/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_i8.json","aliases":["buffer_load_sbyte"]},{"mnemonic":"buffer_load_sbyte","slug":"buffer_load_sbyte","records":1,"summary":"Load 8 bits of signed data from a buffer surface, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_sbyte/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_sbyte.json","aliases":["buffer_load_i8"]},{"mnemonic":"buffer_load_sbyte_d16","slug":"buffer_load_sbyte_d16","records":1,"summary":"Load 8 bits of signed data from a buffer surface, sign extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_sbyte_d16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_sbyte_d16.json","aliases":["buffer_load_d16_i8"]},{"mnemonic":"buffer_load_sbyte_d16_hi","slug":"buffer_load_sbyte_d16_hi","records":1,"summary":"Load 8 bits of signed data from a buffer surface, sign extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_sbyte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_sbyte_d16_hi.json","aliases":["buffer_load_d16_hi_i8"]},{"mnemonic":"buffer_load_short_d16","slug":"buffer_load_short_d16","records":1,"summary":"Load 16 bits of unsigned data from a buffer surface and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_short_d16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_short_d16.json","aliases":["buffer_load_d16_b16"]},{"mnemonic":"buffer_load_short_d16_hi","slug":"buffer_load_short_d16_hi","records":1,"summary":"Load 16 bits of unsigned data from a buffer surface and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_short_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_short_d16_hi.json","aliases":["buffer_load_d16_hi_b16"]},{"mnemonic":"buffer_load_sshort","slug":"buffer_load_sshort","records":1,"summary":"Load 16 bits of signed data from a buffer surface, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_sshort/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_sshort.json","aliases":["buffer_load_i16"]},{"mnemonic":"buffer_load_u16","slug":"buffer_load_u16","records":1,"summary":"Load 16 bits of unsigned data from a buffer surface, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_u16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_u16.json","aliases":["buffer_load_ushort"]},{"mnemonic":"buffer_load_u8","slug":"buffer_load_u8","records":1,"summary":"Load 8 bits of unsigned data from a buffer surface, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_u8/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_u8.json","aliases":["buffer_load_ubyte"]},{"mnemonic":"buffer_load_ubyte","slug":"buffer_load_ubyte","records":1,"summary":"Load 8 bits of unsigned data from a buffer surface, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_ubyte/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_ubyte.json","aliases":["buffer_load_u8"]},{"mnemonic":"buffer_load_ubyte_d16","slug":"buffer_load_ubyte_d16","records":1,"summary":"Load 8 bits of unsigned data from a buffer surface, zero extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_ubyte_d16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_ubyte_d16.json","aliases":["buffer_load_d16_u8"]},{"mnemonic":"buffer_load_ubyte_d16_hi","slug":"buffer_load_ubyte_d16_hi","records":1,"summary":"Load 8 bits of unsigned data from a buffer surface, zero extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_ubyte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_ubyte_d16_hi.json","aliases":["buffer_load_d16_hi_u8"]},{"mnemonic":"buffer_load_ushort","slug":"buffer_load_ushort","records":1,"summary":"Load 16 bits of unsigned data from a buffer surface, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/buffer_load_ushort/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_load_ushort.json","aliases":["buffer_load_u16"]},{"mnemonic":"buffer_store_b128","slug":"buffer_store_b128","records":1,"summary":"Store 128 bits of data from vector input registers into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_b128/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_b128.json","aliases":["buffer_store_dwordx4"]},{"mnemonic":"buffer_store_b16","slug":"buffer_store_b16","records":1,"summary":"Store 16 bits of data from a vector register into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_b16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_b16.json","aliases":["buffer_store_short"]},{"mnemonic":"buffer_store_b32","slug":"buffer_store_b32","records":1,"summary":"Store 32 bits of data from vector input registers into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_b32/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_b32.json","aliases":["buffer_store_dword"]},{"mnemonic":"buffer_store_b64","slug":"buffer_store_b64","records":1,"summary":"Store 64 bits of data from vector input registers into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_b64/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_b64.json","aliases":["buffer_store_dwordx2"]},{"mnemonic":"buffer_store_b8","slug":"buffer_store_b8","records":1,"summary":"Store 8 bits of data from a vector register into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_b8/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_b8.json","aliases":["buffer_store_byte"]},{"mnemonic":"buffer_store_b96","slug":"buffer_store_b96","records":1,"summary":"Store 96 bits of data from vector input registers into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_b96/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_b96.json","aliases":["buffer_store_dwordx3"]},{"mnemonic":"buffer_store_byte","slug":"buffer_store_byte","records":1,"summary":"Store 8 bits of data from a vector register into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_byte/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_byte.json","aliases":["buffer_store_b8"]},{"mnemonic":"buffer_store_byte_d16_hi","slug":"buffer_store_byte_d16_hi","records":1,"summary":"Store 8 bits of data from the high 16 bits of a 32-bit vector register into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_byte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_byte_d16_hi.json","aliases":["buffer_store_d16_hi_b8"]},{"mnemonic":"buffer_store_d16_format_x","slug":"buffer_store_d16_format_x","records":1,"summary":"Convert 16 bits of data from the low 16 bits of a 32-bit vector input register into 1-component formatted data and store the data into a buffer…","page":"https://instructionsets.com/amdgpu/buffer_store_d16_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_d16_format_x.json","aliases":["buffer_store_format_d16_x"]},{"mnemonic":"buffer_store_d16_format_xy","slug":"buffer_store_d16_format_xy","records":1,"summary":"Convert 32 bits of data from vector input registers into 2-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_d16_format_xy/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_d16_format_xy.json","aliases":["buffer_store_format_d16_xy"]},{"mnemonic":"buffer_store_d16_format_xyz","slug":"buffer_store_d16_format_xyz","records":1,"summary":"Convert 48 bits of data from vector input registers into 3-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_d16_format_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_d16_format_xyz.json","aliases":["buffer_store_format_d16_xyz"]},{"mnemonic":"buffer_store_d16_format_xyzw","slug":"buffer_store_d16_format_xyzw","records":1,"summary":"Convert 64 bits of data from vector input registers into 4-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_d16_format_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_d16_format_xyzw.json","aliases":["buffer_store_format_d16_xyzw"]},{"mnemonic":"buffer_store_d16_hi_b16","slug":"buffer_store_d16_hi_b16","records":1,"summary":"Store 16 bits of data from the high 16 bits of a 32-bit vector register into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_d16_hi_b16/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_d16_hi_b16.json","aliases":["buffer_store_short_d16_hi"]},{"mnemonic":"buffer_store_d16_hi_b8","slug":"buffer_store_d16_hi_b8","records":1,"summary":"Store 8 bits of data from the high 16 bits of a 32-bit vector register into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_d16_hi_b8/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_d16_hi_b8.json","aliases":["buffer_store_byte_d16_hi"]},{"mnemonic":"buffer_store_d16_hi_format_x","slug":"buffer_store_d16_hi_format_x","records":1,"summary":"Convert 16 bits of data from the high 16 bits of a 32-bit vector input register into 1-component formatted data and store the data into a buffer…","page":"https://instructionsets.com/amdgpu/buffer_store_d16_hi_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_d16_hi_format_x.json","aliases":["buffer_store_format_d16_hi_x"]},{"mnemonic":"buffer_store_dword","slug":"buffer_store_dword","records":1,"summary":"Store 32 bits of data from vector input registers into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_dword/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_dword.json","aliases":["buffer_store_b32"]},{"mnemonic":"buffer_store_dwordx2","slug":"buffer_store_dwordx2","records":1,"summary":"Store 64 bits of data from vector input registers into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_dwordx2.json","aliases":["buffer_store_b64"]},{"mnemonic":"buffer_store_dwordx3","slug":"buffer_store_dwordx3","records":1,"summary":"Store 96 bits of data from vector input registers into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_dwordx3/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_dwordx3.json","aliases":["buffer_store_b96"]},{"mnemonic":"buffer_store_dwordx4","slug":"buffer_store_dwordx4","records":1,"summary":"Store 128 bits of data from vector input registers into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_dwordx4.json","aliases":["buffer_store_b128"]},{"mnemonic":"buffer_store_format_d16_hi_x","slug":"buffer_store_format_d16_hi_x","records":1,"summary":"Convert 16 bits of data from the high 16 bits of a 32-bit vector input register into 1-component formatted data and store the data into a buffer…","page":"https://instructionsets.com/amdgpu/buffer_store_format_d16_hi_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_format_d16_hi_x.json","aliases":["buffer_store_d16_hi_format_x"]},{"mnemonic":"buffer_store_format_d16_x","slug":"buffer_store_format_d16_x","records":1,"summary":"Convert 16 bits of data from the low 16 bits of a 32-bit vector input register into 1-component formatted data and store the data into a buffer…","page":"https://instructionsets.com/amdgpu/buffer_store_format_d16_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_format_d16_x.json","aliases":["buffer_store_d16_format_x"]},{"mnemonic":"buffer_store_format_d16_xy","slug":"buffer_store_format_d16_xy","records":1,"summary":"Convert 32 bits of data from vector input registers into 2-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_format_d16_xy/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_format_d16_xy.json","aliases":["buffer_store_d16_format_xy"]},{"mnemonic":"buffer_store_format_d16_xyz","slug":"buffer_store_format_d16_xyz","records":1,"summary":"Convert 48 bits of data from vector input registers into 3-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_format_d16_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_format_d16_xyz.json","aliases":["buffer_store_d16_format_xyz"]},{"mnemonic":"buffer_store_format_d16_xyzw","slug":"buffer_store_format_d16_xyzw","records":1,"summary":"Convert 64 bits of data from vector input registers into 4-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_format_d16_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_format_d16_xyzw.json","aliases":["buffer_store_d16_format_xyzw"]},{"mnemonic":"buffer_store_format_x","slug":"buffer_store_format_x","records":1,"summary":"Convert 32 bits of data from vector input registers into 1-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_format_x.json"},{"mnemonic":"buffer_store_format_xy","slug":"buffer_store_format_xy","records":1,"summary":"Convert 64 bits of data from vector input registers into 2-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_format_xy/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_format_xy.json"},{"mnemonic":"buffer_store_format_xyz","slug":"buffer_store_format_xyz","records":1,"summary":"Convert 96 bits of data from vector input registers into 3-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_format_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_format_xyz.json"},{"mnemonic":"buffer_store_format_xyzw","slug":"buffer_store_format_xyzw","records":1,"summary":"Convert 128 bits of data from vector input registers into 4-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_format_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_format_xyzw.json"},{"mnemonic":"buffer_store_lds_dword","slug":"buffer_store_lds_dword","records":1,"summary":"Store one DWORD from LDS memory to system memory without utilizing VGPRs.","page":"https://instructionsets.com/amdgpu/buffer_store_lds_dword/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_lds_dword.json"},{"mnemonic":"buffer_store_short","slug":"buffer_store_short","records":1,"summary":"Store 16 bits of data from a vector register into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_short/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_short.json","aliases":["buffer_store_b16"]},{"mnemonic":"buffer_store_short_d16_hi","slug":"buffer_store_short_d16_hi","records":1,"summary":"Store 16 bits of data from the high 16 bits of a 32-bit vector register into a buffer surface.","page":"https://instructionsets.com/amdgpu/buffer_store_short_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_store_short_d16_hi.json","aliases":["buffer_store_d16_hi_b16"]},{"mnemonic":"buffer_wbinvl1","slug":"buffer_wbinvl1","records":1,"summary":"Write back and invalidate the shader L1. Returns ACK to shader.","page":"https://instructionsets.com/amdgpu/buffer_wbinvl1/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_wbinvl1.json"},{"mnemonic":"buffer_wbinvl1_sc","slug":"buffer_wbinvl1_sc","records":1,"summary":"AMDGPU MUBUF vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/buffer_wbinvl1_sc/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_wbinvl1_sc.json"},{"mnemonic":"buffer_wbinvl1_vol","slug":"buffer_wbinvl1_vol","records":1,"summary":"Write back and invalidate the shader L1 only for lines that are marked volatile. Returns ACK to shader.","page":"https://instructionsets.com/amdgpu/buffer_wbinvl1_vol/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_wbinvl1_vol.json"},{"mnemonic":"buffer_wbl2","slug":"buffer_wbl2","records":1,"summary":"Write back L2 cache. Returns ACK to shader.","page":"https://instructionsets.com/amdgpu/buffer_wbl2/","api":"https://instructionsets.com/api/v1/amdgpu/buffer_wbl2.json"},{"mnemonic":"cluster_load_async_to_lds_b128","slug":"cluster_load_async_to_lds_b128","records":1,"summary":"AMDGPU FLAT vector instruction operating on b128 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/cluster_load_async_to_lds_b128/","api":"https://instructionsets.com/api/v1/amdgpu/cluster_load_async_to_lds_b128.json"},{"mnemonic":"cluster_load_async_to_lds_b32","slug":"cluster_load_async_to_lds_b32","records":1,"summary":"AMDGPU FLAT vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/cluster_load_async_to_lds_b32/","api":"https://instructionsets.com/api/v1/amdgpu/cluster_load_async_to_lds_b32.json"},{"mnemonic":"cluster_load_async_to_lds_b64","slug":"cluster_load_async_to_lds_b64","records":1,"summary":"AMDGPU FLAT vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/cluster_load_async_to_lds_b64/","api":"https://instructionsets.com/api/v1/amdgpu/cluster_load_async_to_lds_b64.json"},{"mnemonic":"cluster_load_async_to_lds_b8","slug":"cluster_load_async_to_lds_b8","records":1,"summary":"AMDGPU FLAT vector instruction operating on b8 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/cluster_load_async_to_lds_b8/","api":"https://instructionsets.com/api/v1/amdgpu/cluster_load_async_to_lds_b8.json"},{"mnemonic":"cluster_load_b128","slug":"cluster_load_b128","records":1,"summary":"AMDGPU FLAT vector instruction operating on b128 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/cluster_load_b128/","api":"https://instructionsets.com/api/v1/amdgpu/cluster_load_b128.json"},{"mnemonic":"cluster_load_b32","slug":"cluster_load_b32","records":1,"summary":"AMDGPU FLAT vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/cluster_load_b32/","api":"https://instructionsets.com/api/v1/amdgpu/cluster_load_b32.json"},{"mnemonic":"cluster_load_b64","slug":"cluster_load_b64","records":1,"summary":"AMDGPU FLAT vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/cluster_load_b64/","api":"https://instructionsets.com/api/v1/amdgpu/cluster_load_b64.json"},{"mnemonic":"ds_add_f32","slug":"ds_add_f32","records":1,"summary":"Add two single-precision float values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_add_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_f32.json"},{"mnemonic":"ds_add_f64","slug":"ds_add_f64","records":1,"summary":"Add a double-precision float value in the data register to a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_add_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_f64.json"},{"mnemonic":"ds_add_gs_reg_rtn","slug":"ds_add_gs_reg_rtn","records":1,"summary":"Perform an atomic add to data in specific registers embedded in GDS rather than operating on GDS memory directly.","page":"https://instructionsets.com/amdgpu/ds_add_gs_reg_rtn/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_gs_reg_rtn.json"},{"mnemonic":"ds_add_rtn_f32","slug":"ds_add_rtn_f32","records":1,"summary":"Add two single-precision float values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_add_rtn_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_rtn_f32.json"},{"mnemonic":"ds_add_rtn_f64","slug":"ds_add_rtn_f64","records":1,"summary":"Add a double-precision float value in the data register to a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_add_rtn_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_rtn_f64.json"},{"mnemonic":"ds_add_rtn_u32","slug":"ds_add_rtn_u32","records":1,"summary":"Add two unsigned 32-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_add_rtn_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_rtn_u32.json"},{"mnemonic":"ds_add_rtn_u64","slug":"ds_add_rtn_u64","records":1,"summary":"Add two unsigned 64-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_add_rtn_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_rtn_u64.json"},{"mnemonic":"ds_add_src2_f32","slug":"ds_add_src2_f32","records":1,"summary":"AMDGPU DS vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_add_src2_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_src2_f32.json"},{"mnemonic":"ds_add_src2_u32","slug":"ds_add_src2_u32","records":1,"summary":"AMDGPU DS vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_add_src2_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_src2_u32.json"},{"mnemonic":"ds_add_src2_u64","slug":"ds_add_src2_u64","records":1,"summary":"AMDGPU DS vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_add_src2_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_src2_u64.json"},{"mnemonic":"ds_add_u32","slug":"ds_add_u32","records":1,"summary":"Atomically add a per-lane value to an LDS location.","page":"https://instructionsets.com/amdgpu/ds_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_u32.json"},{"mnemonic":"ds_add_u64","slug":"ds_add_u64","records":1,"summary":"Add two unsigned 64-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_add_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_add_u64.json"},{"mnemonic":"ds_and_b32","slug":"ds_and_b32","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_and_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_and_b32.json"},{"mnemonic":"ds_and_b64","slug":"ds_and_b64","records":1,"summary":"Calculate bitwise AND given two unsigned 64-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_and_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_and_b64.json"},{"mnemonic":"ds_and_rtn_b32","slug":"ds_and_rtn_b32","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_and_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_and_rtn_b32.json"},{"mnemonic":"ds_and_rtn_b64","slug":"ds_and_rtn_b64","records":1,"summary":"Calculate bitwise AND given two unsigned 64-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_and_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_and_rtn_b64.json"},{"mnemonic":"ds_and_src2_b32","slug":"ds_and_src2_b32","records":1,"summary":"AMDGPU DS vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_and_src2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_and_src2_b32.json"},{"mnemonic":"ds_and_src2_b64","slug":"ds_and_src2_b64","records":1,"summary":"AMDGPU DS vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_and_src2_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_and_src2_b64.json"},{"mnemonic":"ds_append","slug":"ds_append","records":1,"summary":"Add (count_bits(exec_mask)) to the value stored in DS memory at (M0.base + instr_offset) if GDS, or at instr_offset if LDS.","page":"https://instructionsets.com/amdgpu/ds_append/","api":"https://instructionsets.com/api/v1/amdgpu/ds_append.json"},{"mnemonic":"ds_atomic_async_barrier_arrive_b64","slug":"ds_atomic_async_barrier_arrive_b64","records":1,"summary":"AMDGPU DS vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_atomic_async_barrier_arrive_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_atomic_async_barrier_arrive_b64.json"},{"mnemonic":"ds_atomic_barrier_arrive_rtn_b64","slug":"ds_atomic_barrier_arrive_rtn_b64","records":1,"summary":"AMDGPU DS vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_atomic_barrier_arrive_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_atomic_barrier_arrive_rtn_b64.json"},{"mnemonic":"ds_bpermute_b32","slug":"ds_bpermute_b32","records":1,"summary":"Backward permute.","page":"https://instructionsets.com/amdgpu/ds_bpermute_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_bpermute_b32.json"},{"mnemonic":"ds_bpermute_fi_b32","slug":"ds_bpermute_fi_b32","records":1,"summary":"Backward permute and fetch data for invalid lanes.","page":"https://instructionsets.com/amdgpu/ds_bpermute_fi_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_bpermute_fi_b32.json"},{"mnemonic":"ds_bvh_stack_push4_pop1_rtn_b32","slug":"ds_bvh_stack_push4_pop1_rtn_b32","records":1,"summary":"Ray tracing involves traversing a BVH which is a kind of tree where nodes have up to 4 children.","page":"https://instructionsets.com/amdgpu/ds_bvh_stack_push4_pop1_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_bvh_stack_push4_pop1_rtn_b32.json"},{"mnemonic":"ds_bvh_stack_push8_pop1_rtn_b32","slug":"ds_bvh_stack_push8_pop1_rtn_b32","records":1,"summary":"Ray tracing involves traversing a BVH which is a kind of tree where nodes have up to 4 children.","page":"https://instructionsets.com/amdgpu/ds_bvh_stack_push8_pop1_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_bvh_stack_push8_pop1_rtn_b32.json"},{"mnemonic":"ds_bvh_stack_push8_pop2_rtn_b64","slug":"ds_bvh_stack_push8_pop2_rtn_b64","records":1,"summary":"Ray tracing involves traversing a BVH which is a kind of tree where nodes have up to 4 children.","page":"https://instructionsets.com/amdgpu/ds_bvh_stack_push8_pop2_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_bvh_stack_push8_pop2_rtn_b64.json"},{"mnemonic":"ds_bvh_stack_rtn_b32","slug":"ds_bvh_stack_rtn_b32","records":1,"summary":"Ray tracing involves traversing a BVH which is a kind of tree where nodes have up to 4 children.","page":"https://instructionsets.com/amdgpu/ds_bvh_stack_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_bvh_stack_rtn_b32.json"},{"mnemonic":"ds_cmpst_b32","slug":"ds_cmpst_b32","records":1,"summary":"Compare an unsigned 32-bit integer value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpst_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpst_b32.json","aliases":["ds_cmpstore_b32"]},{"mnemonic":"ds_cmpst_b64","slug":"ds_cmpst_b64","records":1,"summary":"Compare an unsigned 64-bit integer value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpst_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpst_b64.json","aliases":["ds_cmpstore_b64"]},{"mnemonic":"ds_cmpst_f32","slug":"ds_cmpst_f32","records":1,"summary":"Compare a single-precision float value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpst_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpst_f32.json","aliases":["ds_cmpstore_f32"]},{"mnemonic":"ds_cmpst_f64","slug":"ds_cmpst_f64","records":1,"summary":"Compare a double-precision float value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpst_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpst_f64.json","aliases":["ds_cmpstore_f64"]},{"mnemonic":"ds_cmpst_rtn_b32","slug":"ds_cmpst_rtn_b32","records":1,"summary":"Compare an unsigned 32-bit integer value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpst_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpst_rtn_b32.json","aliases":["ds_cmpstore_rtn_b32"]},{"mnemonic":"ds_cmpst_rtn_b64","slug":"ds_cmpst_rtn_b64","records":1,"summary":"Compare an unsigned 64-bit integer value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpst_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpst_rtn_b64.json","aliases":["ds_cmpstore_rtn_b64"]},{"mnemonic":"ds_cmpst_rtn_f32","slug":"ds_cmpst_rtn_f32","records":1,"summary":"Compare a single-precision float value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpst_rtn_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpst_rtn_f32.json","aliases":["ds_cmpstore_rtn_f32"]},{"mnemonic":"ds_cmpst_rtn_f64","slug":"ds_cmpst_rtn_f64","records":1,"summary":"Compare a double-precision float value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpst_rtn_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpst_rtn_f64.json","aliases":["ds_cmpstore_rtn_f64"]},{"mnemonic":"ds_cmpstore_b32","slug":"ds_cmpstore_b32","records":1,"summary":"Compare an unsigned 32-bit integer value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpstore_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpstore_b32.json","aliases":["ds_cmpst_b32"]},{"mnemonic":"ds_cmpstore_b64","slug":"ds_cmpstore_b64","records":1,"summary":"Compare an unsigned 64-bit integer value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpstore_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpstore_b64.json","aliases":["ds_cmpst_b64"]},{"mnemonic":"ds_cmpstore_f32","slug":"ds_cmpstore_f32","records":1,"summary":"Compare a single-precision float value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpstore_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpstore_f32.json","aliases":["ds_cmpst_f32"]},{"mnemonic":"ds_cmpstore_f64","slug":"ds_cmpstore_f64","records":1,"summary":"Compare a double-precision float value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpstore_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpstore_f64.json","aliases":["ds_cmpst_f64"]},{"mnemonic":"ds_cmpstore_rtn_b32","slug":"ds_cmpstore_rtn_b32","records":1,"summary":"Compare an unsigned 32-bit integer value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpstore_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpstore_rtn_b32.json","aliases":["ds_cmpst_rtn_b32"]},{"mnemonic":"ds_cmpstore_rtn_b64","slug":"ds_cmpstore_rtn_b64","records":1,"summary":"Compare an unsigned 64-bit integer value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpstore_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpstore_rtn_b64.json","aliases":["ds_cmpst_rtn_b64"]},{"mnemonic":"ds_cmpstore_rtn_f32","slug":"ds_cmpstore_rtn_f32","records":1,"summary":"Compare a single-precision float value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpstore_rtn_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpstore_rtn_f32.json","aliases":["ds_cmpst_rtn_f32"]},{"mnemonic":"ds_cmpstore_rtn_f64","slug":"ds_cmpstore_rtn_f64","records":1,"summary":"Compare a double-precision float value in the data comparison register with a location in a data share, and modify the memory location with a value…","page":"https://instructionsets.com/amdgpu/ds_cmpstore_rtn_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cmpstore_rtn_f64.json","aliases":["ds_cmpst_rtn_f64"]},{"mnemonic":"ds_cond_sub_rtn_u32","slug":"ds_cond_sub_rtn_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value in the data register from a location in a data share only if the memory value is greater than or equal to…","page":"https://instructionsets.com/amdgpu/ds_cond_sub_rtn_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cond_sub_rtn_u32.json"},{"mnemonic":"ds_cond_sub_u32","slug":"ds_cond_sub_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value in the data register from a location in a data share only if the memory value is greater than or equal to…","page":"https://instructionsets.com/amdgpu/ds_cond_sub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_cond_sub_u32.json"},{"mnemonic":"ds_condxchg32_rtn_b64","slug":"ds_condxchg32_rtn_b64","records":1,"summary":"Perform 2 conditional write exchanges, where each conditional write exchange writes a 32 bit value from a data register to a location in data share…","page":"https://instructionsets.com/amdgpu/ds_condxchg32_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_condxchg32_rtn_b64.json"},{"mnemonic":"ds_consume","slug":"ds_consume","records":1,"summary":"Subtract (count_bits(exec_mask)) from the value stored in DS memory at (M0.base + instr_offset) if GDS, or at instr_offset if LDS.","page":"https://instructionsets.com/amdgpu/ds_consume/","api":"https://instructionsets.com/api/v1/amdgpu/ds_consume.json"},{"mnemonic":"ds_dec_rtn_u32","slug":"ds_dec_rtn_u32","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in a data share with wraparound to a value in the data register if the decrement yields a…","page":"https://instructionsets.com/amdgpu/ds_dec_rtn_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_dec_rtn_u32.json"},{"mnemonic":"ds_dec_rtn_u64","slug":"ds_dec_rtn_u64","records":1,"summary":"Decrement an unsigned 64-bit integer value from a location in a data share with wraparound to a value in the data register if the decrement yields a…","page":"https://instructionsets.com/amdgpu/ds_dec_rtn_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_dec_rtn_u64.json"},{"mnemonic":"ds_dec_src2_u32","slug":"ds_dec_src2_u32","records":1,"summary":"AMDGPU DS vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_dec_src2_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_dec_src2_u32.json"},{"mnemonic":"ds_dec_src2_u64","slug":"ds_dec_src2_u64","records":1,"summary":"AMDGPU DS vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_dec_src2_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_dec_src2_u64.json"},{"mnemonic":"ds_dec_u32","slug":"ds_dec_u32","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in a data share with wraparound to a value in the data register if the decrement yields a…","page":"https://instructionsets.com/amdgpu/ds_dec_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_dec_u32.json"},{"mnemonic":"ds_dec_u64","slug":"ds_dec_u64","records":1,"summary":"Decrement an unsigned 64-bit integer value from a location in a data share with wraparound to a value in the data register if the decrement yields a…","page":"https://instructionsets.com/amdgpu/ds_dec_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_dec_u64.json"},{"mnemonic":"ds_direct_load","slug":"ds_direct_load","records":1,"summary":"Read a single 32-bit value from LDS to all lanes.","page":"https://instructionsets.com/amdgpu/ds_direct_load/","api":"https://instructionsets.com/api/v1/amdgpu/ds_direct_load.json","aliases":["lds_direct_load"]},{"mnemonic":"ds_gws_barrier","slug":"ds_gws_barrier","records":1,"summary":"GDS Only: The GWS resource indicated processes this opcode by queueing it until barrier is satisfied.","page":"https://instructionsets.com/amdgpu/ds_gws_barrier/","api":"https://instructionsets.com/api/v1/amdgpu/ds_gws_barrier.json"},{"mnemonic":"ds_gws_init","slug":"ds_gws_init","records":1,"summary":"GDS Only: Initialize a barrier or semaphore resource.","page":"https://instructionsets.com/amdgpu/ds_gws_init/","api":"https://instructionsets.com/api/v1/amdgpu/ds_gws_init.json"},{"mnemonic":"ds_gws_sema_br","slug":"ds_gws_sema_br","records":1,"summary":"GDS Only: The GWS resource indicated processes this opcode by updating the counter by the bulk release delivered count and labeling the resource as a…","page":"https://instructionsets.com/amdgpu/ds_gws_sema_br/","api":"https://instructionsets.com/api/v1/amdgpu/ds_gws_sema_br.json"},{"mnemonic":"ds_gws_sema_p","slug":"ds_gws_sema_p","records":1,"summary":"GDS Only: The GWS resource indicated processes this opcode by queueing it until counter enables a release and then decrementing the counter of the…","page":"https://instructionsets.com/amdgpu/ds_gws_sema_p/","api":"https://instructionsets.com/api/v1/amdgpu/ds_gws_sema_p.json"},{"mnemonic":"ds_gws_sema_release_all","slug":"ds_gws_sema_release_all","records":1,"summary":"GDS Only: The GWS resource (rid) indicated processes this opcode by updating the counter and labeling the specified resource as a semaphore.","page":"https://instructionsets.com/amdgpu/ds_gws_sema_release_all/","api":"https://instructionsets.com/api/v1/amdgpu/ds_gws_sema_release_all.json"},{"mnemonic":"ds_gws_sema_v","slug":"ds_gws_sema_v","records":1,"summary":"GDS Only: The GWS resource indicated processes this opcode by updating the counter and labeling the resource as a semaphore.","page":"https://instructionsets.com/amdgpu/ds_gws_sema_v/","api":"https://instructionsets.com/api/v1/amdgpu/ds_gws_sema_v.json"},{"mnemonic":"ds_inc_rtn_u32","slug":"ds_inc_rtn_u32","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in a data share with wraparound to 0 if the value exceeds a value in the data register.","page":"https://instructionsets.com/amdgpu/ds_inc_rtn_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_inc_rtn_u32.json"},{"mnemonic":"ds_inc_rtn_u64","slug":"ds_inc_rtn_u64","records":1,"summary":"Increment an unsigned 64-bit integer value from a location in a data share with wraparound to 0 if the value exceeds a value in the data register.","page":"https://instructionsets.com/amdgpu/ds_inc_rtn_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_inc_rtn_u64.json"},{"mnemonic":"ds_inc_src2_u32","slug":"ds_inc_src2_u32","records":1,"summary":"AMDGPU DS vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_inc_src2_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_inc_src2_u32.json"},{"mnemonic":"ds_inc_src2_u64","slug":"ds_inc_src2_u64","records":1,"summary":"AMDGPU DS vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_inc_src2_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_inc_src2_u64.json"},{"mnemonic":"ds_inc_u32","slug":"ds_inc_u32","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in a data share with wraparound to 0 if the value exceeds a value in the data register.","page":"https://instructionsets.com/amdgpu/ds_inc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_inc_u32.json"},{"mnemonic":"ds_inc_u64","slug":"ds_inc_u64","records":1,"summary":"Increment an unsigned 64-bit integer value from a location in a data share with wraparound to 0 if the value exceeds a value in the data register.","page":"https://instructionsets.com/amdgpu/ds_inc_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_inc_u64.json"},{"mnemonic":"ds_load_2addr_b32","slug":"ds_load_2addr_b32","records":1,"summary":"Load 32 bits of data from one location in a data share and then 32 bits of data from a second location in a data share and store the results into a…","page":"https://instructionsets.com/amdgpu/ds_load_2addr_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_2addr_b32.json","aliases":["ds_read2_b32"]},{"mnemonic":"ds_load_2addr_b64","slug":"ds_load_2addr_b64","records":1,"summary":"Load 64 bits of data from one location in a data share and then 64 bits of data from a second location in a data share and store the results into a…","page":"https://instructionsets.com/amdgpu/ds_load_2addr_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_2addr_b64.json","aliases":["ds_read2_b64"]},{"mnemonic":"ds_load_2addr_stride64_b32","slug":"ds_load_2addr_stride64_b32","records":1,"summary":"Load 32 bits of data from one location in a data share and then 32 bits of data from a second location in a data share and store the results into a…","page":"https://instructionsets.com/amdgpu/ds_load_2addr_stride64_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_2addr_stride64_b32.json","aliases":["ds_read2st64_b32"]},{"mnemonic":"ds_load_2addr_stride64_b64","slug":"ds_load_2addr_stride64_b64","records":1,"summary":"Load 64 bits of data from one location in a data share and then 64 bits of data from a second location in a data share and store the results into a…","page":"https://instructionsets.com/amdgpu/ds_load_2addr_stride64_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_2addr_stride64_b64.json","aliases":["ds_read2st64_b64"]},{"mnemonic":"ds_load_addtid_b32","slug":"ds_load_addtid_b32","records":1,"summary":"Load 32 bits of data from a data share into a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_addtid_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_addtid_b32.json","aliases":["ds_read_addtid_b32"]},{"mnemonic":"ds_load_b128","slug":"ds_load_b128","records":1,"summary":"Load 128 bits of data from a data share into a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_b128/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_b128.json","aliases":["ds_read_b128"]},{"mnemonic":"ds_load_b32","slug":"ds_load_b32","records":1,"summary":"Load 32 bits of data from a data share into a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_b32.json","aliases":["ds_read_b32"]},{"mnemonic":"ds_load_b64","slug":"ds_load_b64","records":1,"summary":"Load 64 bits of data from a data share into a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_b64.json","aliases":["ds_read_b64"]},{"mnemonic":"ds_load_b96","slug":"ds_load_b96","records":1,"summary":"Load 96 bits of data from a data share into a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_b96/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_b96.json","aliases":["ds_read_b96"]},{"mnemonic":"ds_load_i16","slug":"ds_load_i16","records":1,"summary":"Load 16 bits of signed data from a data share, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_i16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_i16.json","aliases":["ds_read_i16"]},{"mnemonic":"ds_load_i8","slug":"ds_load_i8","records":1,"summary":"Load 8 bits of signed data from a data share, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_i8/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_i8.json","aliases":["ds_read_i8"]},{"mnemonic":"ds_load_i8_d16","slug":"ds_load_i8_d16","records":1,"summary":"Load 8 bits of signed data from a data share, sign extend to 16 bits and store the result into the low 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_i8_d16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_i8_d16.json","aliases":["ds_read_i8_d16"]},{"mnemonic":"ds_load_i8_d16_hi","slug":"ds_load_i8_d16_hi","records":1,"summary":"Load 8 bits of signed data from a data share, sign extend to 16 bits and store the result into the high 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_i8_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_i8_d16_hi.json","aliases":["ds_read_i8_d16_hi"]},{"mnemonic":"ds_load_tr16_b128","slug":"ds_load_tr16_b128","records":1,"summary":"AMDGPU DS vector instruction operating on b128 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_load_tr16_b128/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_tr16_b128.json","aliases":["ds_load_b128_tr_b16","ds_load_tr_b128"]},{"mnemonic":"ds_load_tr4_b64","slug":"ds_load_tr4_b64","records":1,"summary":"AMDGPU DS vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_load_tr4_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_tr4_b64.json","aliases":["ds_load_b64_tr_b4"]},{"mnemonic":"ds_load_tr6_b96","slug":"ds_load_tr6_b96","records":1,"summary":"AMDGPU DS vector instruction operating on b96 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_load_tr6_b96/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_tr6_b96.json","aliases":["ds_load_b128_tr_b6"]},{"mnemonic":"ds_load_tr8_b64","slug":"ds_load_tr8_b64","records":1,"summary":"AMDGPU DS vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_load_tr8_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_tr8_b64.json","aliases":["ds_load_b64_tr_b8","ds_load_tr_b64"]},{"mnemonic":"ds_load_u16","slug":"ds_load_u16","records":1,"summary":"Load 16 bits of unsigned data from a data share, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_u16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_u16.json","aliases":["ds_read_u16"]},{"mnemonic":"ds_load_u16_d16","slug":"ds_load_u16_d16","records":1,"summary":"Load 16 bits of unsigned data from a data share and store the result into the low 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_u16_d16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_u16_d16.json","aliases":["ds_read_u16_d16"]},{"mnemonic":"ds_load_u16_d16_hi","slug":"ds_load_u16_d16_hi","records":1,"summary":"Load 16 bits of unsigned data from a data share and store the result into the high 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_u16_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_u16_d16_hi.json","aliases":["ds_read_u16_d16_hi"]},{"mnemonic":"ds_load_u8","slug":"ds_load_u8","records":1,"summary":"Load 8 bits of unsigned data from a data share, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_u8/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_u8.json","aliases":["ds_read_u8"]},{"mnemonic":"ds_load_u8_d16","slug":"ds_load_u8_d16","records":1,"summary":"Load 8 bits of unsigned data from a data share, zero extend to 16 bits and store the result into the low 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_u8_d16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_u8_d16.json","aliases":["ds_read_u8_d16"]},{"mnemonic":"ds_load_u8_d16_hi","slug":"ds_load_u8_d16_hi","records":1,"summary":"Load 8 bits of unsigned data from a data share, zero extend to 16 bits and store the result into the high 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_load_u8_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_load_u8_d16_hi.json","aliases":["ds_read_u8_d16_hi"]},{"mnemonic":"ds_max_f32","slug":"ds_max_f32","records":1,"summary":"Select the maximum of two single-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_f32.json","aliases":["ds_max_num_f32"]},{"mnemonic":"ds_max_f64","slug":"ds_max_f64","records":1,"summary":"Select the maximum of two double-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_f64.json","aliases":["ds_max_num_f64"]},{"mnemonic":"ds_max_i32","slug":"ds_max_i32","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_i32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_i32.json"},{"mnemonic":"ds_max_i64","slug":"ds_max_i64","records":1,"summary":"Select the maximum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_i64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_i64.json"},{"mnemonic":"ds_max_num_f32","slug":"ds_max_num_f32","records":1,"summary":"Select the IEEE maximumNumber() of two single-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_num_f32.json","aliases":["ds_max_f32"]},{"mnemonic":"ds_max_num_f64","slug":"ds_max_num_f64","records":1,"summary":"Select the IEEE maximumNumber() of two double-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_num_f64.json","aliases":["ds_max_f64"]},{"mnemonic":"ds_max_num_rtn_f32","slug":"ds_max_num_rtn_f32","records":1,"summary":"Select the IEEE maximumNumber() of two single-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_num_rtn_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_num_rtn_f32.json","aliases":["ds_max_rtn_f32"]},{"mnemonic":"ds_max_num_rtn_f64","slug":"ds_max_num_rtn_f64","records":1,"summary":"Select the IEEE maximumNumber() of two double-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_num_rtn_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_num_rtn_f64.json","aliases":["ds_max_rtn_f64"]},{"mnemonic":"ds_max_rtn_f32","slug":"ds_max_rtn_f32","records":1,"summary":"Select the maximum of two single-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_rtn_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_rtn_f32.json","aliases":["ds_max_num_rtn_f32"]},{"mnemonic":"ds_max_rtn_f64","slug":"ds_max_rtn_f64","records":1,"summary":"Select the maximum of two double-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_rtn_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_rtn_f64.json","aliases":["ds_max_num_rtn_f64"]},{"mnemonic":"ds_max_rtn_i32","slug":"ds_max_rtn_i32","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_rtn_i32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_rtn_i32.json"},{"mnemonic":"ds_max_rtn_i64","slug":"ds_max_rtn_i64","records":1,"summary":"Select the maximum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_rtn_i64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_rtn_i64.json"},{"mnemonic":"ds_max_rtn_u32","slug":"ds_max_rtn_u32","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_rtn_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_rtn_u32.json"},{"mnemonic":"ds_max_rtn_u64","slug":"ds_max_rtn_u64","records":1,"summary":"Select the maximum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_rtn_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_rtn_u64.json"},{"mnemonic":"ds_max_src2_f32","slug":"ds_max_src2_f32","records":1,"summary":"AMDGPU DS vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_max_src2_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_src2_f32.json"},{"mnemonic":"ds_max_src2_f64","slug":"ds_max_src2_f64","records":1,"summary":"AMDGPU DS vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_max_src2_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_src2_f64.json"},{"mnemonic":"ds_max_src2_i32","slug":"ds_max_src2_i32","records":1,"summary":"AMDGPU DS vector instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_max_src2_i32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_src2_i32.json"},{"mnemonic":"ds_max_src2_i64","slug":"ds_max_src2_i64","records":1,"summary":"AMDGPU DS vector instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_max_src2_i64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_src2_i64.json"},{"mnemonic":"ds_max_src2_u32","slug":"ds_max_src2_u32","records":1,"summary":"AMDGPU DS vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_max_src2_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_src2_u32.json"},{"mnemonic":"ds_max_src2_u64","slug":"ds_max_src2_u64","records":1,"summary":"AMDGPU DS vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_max_src2_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_src2_u64.json"},{"mnemonic":"ds_max_u32","slug":"ds_max_u32","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_u32.json"},{"mnemonic":"ds_max_u64","slug":"ds_max_u64","records":1,"summary":"Select the maximum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_max_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_max_u64.json"},{"mnemonic":"ds_min_f32","slug":"ds_min_f32","records":1,"summary":"Select the minimum of two single-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_f32.json","aliases":["ds_min_num_f32"]},{"mnemonic":"ds_min_f64","slug":"ds_min_f64","records":1,"summary":"Select the minimum of two double-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_f64.json","aliases":["ds_min_num_f64"]},{"mnemonic":"ds_min_i32","slug":"ds_min_i32","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_i32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_i32.json"},{"mnemonic":"ds_min_i64","slug":"ds_min_i64","records":1,"summary":"Select the minimum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_i64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_i64.json"},{"mnemonic":"ds_min_num_f32","slug":"ds_min_num_f32","records":1,"summary":"Select the IEEE minimumNumber() of two single-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_num_f32.json","aliases":["ds_min_f32"]},{"mnemonic":"ds_min_num_f64","slug":"ds_min_num_f64","records":1,"summary":"Select the IEEE minimumNumber() of two double-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_num_f64.json","aliases":["ds_min_f64"]},{"mnemonic":"ds_min_num_rtn_f32","slug":"ds_min_num_rtn_f32","records":1,"summary":"Select the IEEE minimumNumber() of two single-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_num_rtn_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_num_rtn_f32.json","aliases":["ds_min_rtn_f32"]},{"mnemonic":"ds_min_num_rtn_f64","slug":"ds_min_num_rtn_f64","records":1,"summary":"Select the IEEE minimumNumber() of two double-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_num_rtn_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_num_rtn_f64.json","aliases":["ds_min_rtn_f64"]},{"mnemonic":"ds_min_rtn_f32","slug":"ds_min_rtn_f32","records":1,"summary":"Select the minimum of two single-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_rtn_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_rtn_f32.json","aliases":["ds_min_num_rtn_f32"]},{"mnemonic":"ds_min_rtn_f64","slug":"ds_min_rtn_f64","records":1,"summary":"Select the minimum of two double-precision float inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_rtn_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_rtn_f64.json","aliases":["ds_min_num_rtn_f64"]},{"mnemonic":"ds_min_rtn_i32","slug":"ds_min_rtn_i32","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_rtn_i32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_rtn_i32.json"},{"mnemonic":"ds_min_rtn_i64","slug":"ds_min_rtn_i64","records":1,"summary":"Select the minimum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_rtn_i64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_rtn_i64.json"},{"mnemonic":"ds_min_rtn_u32","slug":"ds_min_rtn_u32","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_rtn_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_rtn_u32.json"},{"mnemonic":"ds_min_rtn_u64","slug":"ds_min_rtn_u64","records":1,"summary":"Select the minimum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_rtn_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_rtn_u64.json"},{"mnemonic":"ds_min_src2_f32","slug":"ds_min_src2_f32","records":1,"summary":"AMDGPU DS vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_min_src2_f32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_src2_f32.json"},{"mnemonic":"ds_min_src2_f64","slug":"ds_min_src2_f64","records":1,"summary":"AMDGPU DS vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_min_src2_f64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_src2_f64.json"},{"mnemonic":"ds_min_src2_i32","slug":"ds_min_src2_i32","records":1,"summary":"AMDGPU DS vector instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_min_src2_i32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_src2_i32.json"},{"mnemonic":"ds_min_src2_i64","slug":"ds_min_src2_i64","records":1,"summary":"AMDGPU DS vector instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_min_src2_i64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_src2_i64.json"},{"mnemonic":"ds_min_src2_u32","slug":"ds_min_src2_u32","records":1,"summary":"AMDGPU DS vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_min_src2_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_src2_u32.json"},{"mnemonic":"ds_min_src2_u64","slug":"ds_min_src2_u64","records":1,"summary":"AMDGPU DS vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_min_src2_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_src2_u64.json"},{"mnemonic":"ds_min_u32","slug":"ds_min_u32","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_u32.json"},{"mnemonic":"ds_min_u64","slug":"ds_min_u64","records":1,"summary":"Select the minimum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_min_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_min_u64.json"},{"mnemonic":"ds_mskor_b32","slug":"ds_mskor_b32","records":1,"summary":"Calculate masked bitwise OR on an unsigned 32-bit integer location in a data share, given mask value and bits to OR in the data registers.","page":"https://instructionsets.com/amdgpu/ds_mskor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_mskor_b32.json"},{"mnemonic":"ds_mskor_b64","slug":"ds_mskor_b64","records":1,"summary":"Calculate masked bitwise OR on an unsigned 64-bit integer location in a data share, given mask value and bits to OR in the data registers.","page":"https://instructionsets.com/amdgpu/ds_mskor_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_mskor_b64.json"},{"mnemonic":"ds_mskor_rtn_b32","slug":"ds_mskor_rtn_b32","records":1,"summary":"Calculate masked bitwise OR on an unsigned 32-bit integer location in a data share, given mask value and bits to OR in the data registers.","page":"https://instructionsets.com/amdgpu/ds_mskor_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_mskor_rtn_b32.json"},{"mnemonic":"ds_mskor_rtn_b64","slug":"ds_mskor_rtn_b64","records":1,"summary":"Calculate masked bitwise OR on an unsigned 64-bit integer location in a data share, given mask value and bits to OR in the data registers.","page":"https://instructionsets.com/amdgpu/ds_mskor_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_mskor_rtn_b64.json"},{"mnemonic":"ds_nop","slug":"ds_nop","records":1,"summary":"Do nothing.","page":"https://instructionsets.com/amdgpu/ds_nop/","api":"https://instructionsets.com/api/v1/amdgpu/ds_nop.json"},{"mnemonic":"ds_or_b32","slug":"ds_or_b32","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_or_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_or_b32.json"},{"mnemonic":"ds_or_b64","slug":"ds_or_b64","records":1,"summary":"Calculate bitwise OR given two unsigned 64-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_or_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_or_b64.json"},{"mnemonic":"ds_or_rtn_b32","slug":"ds_or_rtn_b32","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_or_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_or_rtn_b32.json"},{"mnemonic":"ds_or_rtn_b64","slug":"ds_or_rtn_b64","records":1,"summary":"Calculate bitwise OR given two unsigned 64-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_or_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_or_rtn_b64.json"},{"mnemonic":"ds_or_src2_b32","slug":"ds_or_src2_b32","records":1,"summary":"AMDGPU DS vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_or_src2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_or_src2_b32.json"},{"mnemonic":"ds_or_src2_b64","slug":"ds_or_src2_b64","records":1,"summary":"AMDGPU DS vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_or_src2_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_or_src2_b64.json"},{"mnemonic":"ds_ordered_count","slug":"ds_ordered_count","records":1,"summary":"GDS-only.","page":"https://instructionsets.com/amdgpu/ds_ordered_count/","api":"https://instructionsets.com/api/v1/amdgpu/ds_ordered_count.json"},{"mnemonic":"ds_param_load","slug":"ds_param_load","records":1,"summary":"Transfer parameter data from LDS to VGPRs and expand data in LDS using the NewPrimMask (provided in M0) to place per-quad data into lanes 0-3 of each…","page":"https://instructionsets.com/amdgpu/ds_param_load/","api":"https://instructionsets.com/api/v1/amdgpu/ds_param_load.json","aliases":["lds_param_load"]},{"mnemonic":"ds_permute_b32","slug":"ds_permute_b32","records":1,"summary":"Forward-permute: each lane sends its value to a lane index computed by another lane, via the LDS crossbar (no LDS storage consumed).","page":"https://instructionsets.com/amdgpu/ds_permute_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_permute_b32.json"},{"mnemonic":"ds_pk_add_bf16","slug":"ds_pk_add_bf16","records":1,"summary":"Add a packed 2-component BF16 float value in the data register to a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_pk_add_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_pk_add_bf16.json"},{"mnemonic":"ds_pk_add_f16","slug":"ds_pk_add_f16","records":1,"summary":"Add a packed 2-component half-precision float value in the data register to a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_pk_add_f16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_pk_add_f16.json"},{"mnemonic":"ds_pk_add_rtn_bf16","slug":"ds_pk_add_rtn_bf16","records":1,"summary":"Add a packed 2-component BF16 float value in the data register to a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_pk_add_rtn_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_pk_add_rtn_bf16.json"},{"mnemonic":"ds_pk_add_rtn_f16","slug":"ds_pk_add_rtn_f16","records":1,"summary":"Add a packed 2-component half-precision float value in the data register to a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_pk_add_rtn_f16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_pk_add_rtn_f16.json"},{"mnemonic":"ds_read2_b32","slug":"ds_read2_b32","records":1,"summary":"Load 32 bits of data from one location in a data share and then 32 bits of data from a second location in a data share and store the results into a…","page":"https://instructionsets.com/amdgpu/ds_read2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read2_b32.json","aliases":["ds_load_2addr_b32"]},{"mnemonic":"ds_read2_b64","slug":"ds_read2_b64","records":1,"summary":"Load 64 bits of data from one location in a data share and then 64 bits of data from a second location in a data share and store the results into a…","page":"https://instructionsets.com/amdgpu/ds_read2_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read2_b64.json","aliases":["ds_load_2addr_b64"]},{"mnemonic":"ds_read2st64_b32","slug":"ds_read2st64_b32","records":1,"summary":"Load 32 bits of data from one location in a data share and then 32 bits of data from a second location in a data share and store the results into a…","page":"https://instructionsets.com/amdgpu/ds_read2st64_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read2st64_b32.json","aliases":["ds_load_2addr_stride64_b32"]},{"mnemonic":"ds_read2st64_b64","slug":"ds_read2st64_b64","records":1,"summary":"Load 64 bits of data from one location in a data share and then 64 bits of data from a second location in a data share and store the results into a…","page":"https://instructionsets.com/amdgpu/ds_read2st64_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read2st64_b64.json","aliases":["ds_load_2addr_stride64_b64"]},{"mnemonic":"ds_read_addtid_b32","slug":"ds_read_addtid_b32","records":1,"summary":"Load 32 bits of data from a data share into a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_addtid_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_addtid_b32.json","aliases":["ds_load_addtid_b32"]},{"mnemonic":"ds_read_b128","slug":"ds_read_b128","records":1,"summary":"Load 128 bits of data from a data share into a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_b128/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_b128.json","aliases":["ds_load_b128"]},{"mnemonic":"ds_read_b32","slug":"ds_read_b32","records":1,"summary":"Read one 32-bit value per lane from the Local Data Share (LDS).","page":"https://instructionsets.com/amdgpu/ds_read_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_b32.json","aliases":["ds_load_b32"]},{"mnemonic":"ds_read_b64","slug":"ds_read_b64","records":1,"summary":"Load 64 bits of data from a data share into a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_b64.json","aliases":["ds_load_b64"]},{"mnemonic":"ds_read_b64_tr_b16","slug":"ds_read_b64_tr_b16","records":1,"summary":"Read 64 bits of data per lane from data share.","page":"https://instructionsets.com/amdgpu/ds_read_b64_tr_b16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_b64_tr_b16.json"},{"mnemonic":"ds_read_b64_tr_b4","slug":"ds_read_b64_tr_b4","records":1,"summary":"Read 64 bits of data per lane from data share.","page":"https://instructionsets.com/amdgpu/ds_read_b64_tr_b4/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_b64_tr_b4.json"},{"mnemonic":"ds_read_b64_tr_b8","slug":"ds_read_b64_tr_b8","records":1,"summary":"Read 64 bits of data per lane from data share.","page":"https://instructionsets.com/amdgpu/ds_read_b64_tr_b8/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_b64_tr_b8.json"},{"mnemonic":"ds_read_b96","slug":"ds_read_b96","records":1,"summary":"Load 96 bits of data from a data share into a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_b96/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_b96.json","aliases":["ds_load_b96"]},{"mnemonic":"ds_read_b96_tr_b6","slug":"ds_read_b96_tr_b6","records":1,"summary":"Read 96 bits of data per lane from data share.","page":"https://instructionsets.com/amdgpu/ds_read_b96_tr_b6/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_b96_tr_b6.json","aliases":["ds_read_b128_tr_b6"]},{"mnemonic":"ds_read_i16","slug":"ds_read_i16","records":1,"summary":"Load 16 bits of signed data from a data share, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_i16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_i16.json","aliases":["ds_load_i16"]},{"mnemonic":"ds_read_i8","slug":"ds_read_i8","records":1,"summary":"Load 8 bits of signed data from a data share, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_i8/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_i8.json","aliases":["ds_load_i8"]},{"mnemonic":"ds_read_i8_d16","slug":"ds_read_i8_d16","records":1,"summary":"Load 8 bits of signed data from a data share, sign extend to 16 bits and store the result into the low 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_i8_d16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_i8_d16.json","aliases":["ds_load_i8_d16"]},{"mnemonic":"ds_read_i8_d16_hi","slug":"ds_read_i8_d16_hi","records":1,"summary":"Load 8 bits of signed data from a data share, sign extend to 16 bits and store the result into the high 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_i8_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_i8_d16_hi.json","aliases":["ds_load_i8_d16_hi"]},{"mnemonic":"ds_read_u16","slug":"ds_read_u16","records":1,"summary":"Load 16 bits of unsigned data from a data share, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_u16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_u16.json","aliases":["ds_load_u16"]},{"mnemonic":"ds_read_u16_d16","slug":"ds_read_u16_d16","records":1,"summary":"Load 16 bits of unsigned data from a data share and store the result into the low 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_u16_d16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_u16_d16.json","aliases":["ds_load_u16_d16"]},{"mnemonic":"ds_read_u16_d16_hi","slug":"ds_read_u16_d16_hi","records":1,"summary":"Load 16 bits of unsigned data from a data share and store the result into the high 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_u16_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_u16_d16_hi.json","aliases":["ds_load_u16_d16_hi"]},{"mnemonic":"ds_read_u8","slug":"ds_read_u8","records":1,"summary":"Load 8 bits of unsigned data from a data share, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_u8/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_u8.json","aliases":["ds_load_u8"]},{"mnemonic":"ds_read_u8_d16","slug":"ds_read_u8_d16","records":1,"summary":"Load 8 bits of unsigned data from a data share, zero extend to 16 bits and store the result into the low 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_u8_d16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_u8_d16.json","aliases":["ds_load_u8_d16"]},{"mnemonic":"ds_read_u8_d16_hi","slug":"ds_read_u8_d16_hi","records":1,"summary":"Load 8 bits of unsigned data from a data share, zero extend to 16 bits and store the result into the high 16 bits of a vector register.","page":"https://instructionsets.com/amdgpu/ds_read_u8_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_read_u8_d16_hi.json","aliases":["ds_load_u8_d16_hi"]},{"mnemonic":"ds_rsub_rtn_u32","slug":"ds_rsub_rtn_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in a location in a data share from a value stored in the data register.","page":"https://instructionsets.com/amdgpu/ds_rsub_rtn_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_rsub_rtn_u32.json","aliases":["ds_subrev_rtn_u32"]},{"mnemonic":"ds_rsub_rtn_u64","slug":"ds_rsub_rtn_u64","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in a location in a data share from a value stored in the data register.","page":"https://instructionsets.com/amdgpu/ds_rsub_rtn_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_rsub_rtn_u64.json","aliases":["ds_subrev_rtn_u64"]},{"mnemonic":"ds_rsub_src2_u32","slug":"ds_rsub_src2_u32","records":1,"summary":"AMDGPU DS vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_rsub_src2_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_rsub_src2_u32.json"},{"mnemonic":"ds_rsub_src2_u64","slug":"ds_rsub_src2_u64","records":1,"summary":"AMDGPU DS vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_rsub_src2_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_rsub_src2_u64.json"},{"mnemonic":"ds_rsub_u32","slug":"ds_rsub_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in a location in a data share from a value stored in the data register.","page":"https://instructionsets.com/amdgpu/ds_rsub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_rsub_u32.json","aliases":["ds_subrev_u32"]},{"mnemonic":"ds_rsub_u64","slug":"ds_rsub_u64","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in a location in a data share from a value stored in the data register.","page":"https://instructionsets.com/amdgpu/ds_rsub_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_rsub_u64.json","aliases":["ds_subrev_u64"]},{"mnemonic":"ds_store_2addr_b32","slug":"ds_store_2addr_b32","records":1,"summary":"Store 32 bits of data from one vector input register and then 32 bits of data from a second vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_2addr_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_2addr_b32.json","aliases":["ds_write2_b32"]},{"mnemonic":"ds_store_2addr_b64","slug":"ds_store_2addr_b64","records":1,"summary":"Store 64 bits of data from one vector input register and then 64 bits of data from a second vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_2addr_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_2addr_b64.json","aliases":["ds_write2_b64"]},{"mnemonic":"ds_store_2addr_stride64_b32","slug":"ds_store_2addr_stride64_b32","records":1,"summary":"Store 32 bits of data from one vector input register and then 32 bits of data from a second vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_2addr_stride64_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_2addr_stride64_b32.json","aliases":["ds_write2st64_b32"]},{"mnemonic":"ds_store_2addr_stride64_b64","slug":"ds_store_2addr_stride64_b64","records":1,"summary":"Store 64 bits of data from one vector input register and then 64 bits of data from a second vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_2addr_stride64_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_2addr_stride64_b64.json","aliases":["ds_write2st64_b64"]},{"mnemonic":"ds_store_addtid_b32","slug":"ds_store_addtid_b32","records":1,"summary":"Store 32 bits of data from a vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_addtid_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_addtid_b32.json","aliases":["ds_write_addtid_b32"]},{"mnemonic":"ds_store_b128","slug":"ds_store_b128","records":1,"summary":"Store 128 bits of data from a vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_b128/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_b128.json","aliases":["ds_write_b128"]},{"mnemonic":"ds_store_b16","slug":"ds_store_b16","records":1,"summary":"Store 16 bits of data from a vector register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_b16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_b16.json","aliases":["ds_write_b16"]},{"mnemonic":"ds_store_b16_d16_hi","slug":"ds_store_b16_d16_hi","records":1,"summary":"Store 16 bits of data from the high bits of a vector register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_b16_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_b16_d16_hi.json","aliases":["ds_write_b16_d16_hi"]},{"mnemonic":"ds_store_b32","slug":"ds_store_b32","records":1,"summary":"Store 32 bits of data from a vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_b32.json","aliases":["ds_write_b32"]},{"mnemonic":"ds_store_b64","slug":"ds_store_b64","records":1,"summary":"Store 64 bits of data from a vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_b64.json","aliases":["ds_write_b64"]},{"mnemonic":"ds_store_b8","slug":"ds_store_b8","records":1,"summary":"Store 8 bits of data from a vector register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_b8/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_b8.json","aliases":["ds_write_b8"]},{"mnemonic":"ds_store_b8_d16_hi","slug":"ds_store_b8_d16_hi","records":1,"summary":"Store 8 bits of data from the high bits of a vector register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_b8_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_b8_d16_hi.json","aliases":["ds_write_b8_d16_hi"]},{"mnemonic":"ds_store_b96","slug":"ds_store_b96","records":1,"summary":"Store 96 bits of data from a vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_store_b96/","api":"https://instructionsets.com/api/v1/amdgpu/ds_store_b96.json","aliases":["ds_write_b96"]},{"mnemonic":"ds_storexchg_2addr_rtn_b32","slug":"ds_storexchg_2addr_rtn_b32","records":1,"summary":"Swap two unsigned 32-bit integer values in the data registers with two locations in a data share.","page":"https://instructionsets.com/amdgpu/ds_storexchg_2addr_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_storexchg_2addr_rtn_b32.json","aliases":["ds_wrxchg2_rtn_b32"]},{"mnemonic":"ds_storexchg_2addr_rtn_b64","slug":"ds_storexchg_2addr_rtn_b64","records":1,"summary":"Swap two unsigned 64-bit integer values in the data registers with two locations in a data share.","page":"https://instructionsets.com/amdgpu/ds_storexchg_2addr_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_storexchg_2addr_rtn_b64.json","aliases":["ds_wrxchg2_rtn_b64"]},{"mnemonic":"ds_storexchg_2addr_stride64_rtn_b32","slug":"ds_storexchg_2addr_stride64_rtn_b32","records":1,"summary":"Swap two unsigned 32-bit integer values in the data registers with two locations in a data share.","page":"https://instructionsets.com/amdgpu/ds_storexchg_2addr_stride64_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_storexchg_2addr_stride64_rtn_b32.json","aliases":["ds_wrxchg2st64_rtn_b32"]},{"mnemonic":"ds_storexchg_2addr_stride64_rtn_b64","slug":"ds_storexchg_2addr_stride64_rtn_b64","records":1,"summary":"Swap two unsigned 64-bit integer values in the data registers with two locations in a data share.","page":"https://instructionsets.com/amdgpu/ds_storexchg_2addr_stride64_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_storexchg_2addr_stride64_rtn_b64.json","aliases":["ds_wrxchg2st64_rtn_b64"]},{"mnemonic":"ds_storexchg_rtn_b32","slug":"ds_storexchg_rtn_b32","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_storexchg_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_storexchg_rtn_b32.json","aliases":["ds_wrxchg_rtn_b32"]},{"mnemonic":"ds_storexchg_rtn_b64","slug":"ds_storexchg_rtn_b64","records":1,"summary":"Swap an unsigned 64-bit integer value in the data register with a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_storexchg_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_storexchg_rtn_b64.json","aliases":["ds_wrxchg_rtn_b64"]},{"mnemonic":"ds_sub_clamp_rtn_u32","slug":"ds_sub_clamp_rtn_u32","records":1,"summary":"Subtract an unsigned 32-bit integer location in a data share from a value in the data register and clamp the result to zero.","page":"https://instructionsets.com/amdgpu/ds_sub_clamp_rtn_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_sub_clamp_rtn_u32.json"},{"mnemonic":"ds_sub_clamp_u32","slug":"ds_sub_clamp_u32","records":1,"summary":"Subtract an unsigned 32-bit integer location in a data share from a value in the data register and clamp the result to zero.","page":"https://instructionsets.com/amdgpu/ds_sub_clamp_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_sub_clamp_u32.json"},{"mnemonic":"ds_sub_gs_reg_rtn","slug":"ds_sub_gs_reg_rtn","records":1,"summary":"Perform an atomic subtraction from data in specific registers embedded in GDS rather than operating on GDS memory directly.","page":"https://instructionsets.com/amdgpu/ds_sub_gs_reg_rtn/","api":"https://instructionsets.com/api/v1/amdgpu/ds_sub_gs_reg_rtn.json"},{"mnemonic":"ds_sub_rtn_u32","slug":"ds_sub_rtn_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_sub_rtn_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_sub_rtn_u32.json"},{"mnemonic":"ds_sub_rtn_u64","slug":"ds_sub_rtn_u64","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in the data register from a value stored in a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_sub_rtn_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_sub_rtn_u64.json"},{"mnemonic":"ds_sub_src2_u32","slug":"ds_sub_src2_u32","records":1,"summary":"AMDGPU DS vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_sub_src2_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_sub_src2_u32.json"},{"mnemonic":"ds_sub_src2_u64","slug":"ds_sub_src2_u64","records":1,"summary":"AMDGPU DS vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_sub_src2_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_sub_src2_u64.json"},{"mnemonic":"ds_sub_u32","slug":"ds_sub_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_sub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_sub_u32.json"},{"mnemonic":"ds_sub_u64","slug":"ds_sub_u64","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in the data register from a value stored in a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_sub_u64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_sub_u64.json"},{"mnemonic":"ds_swizzle_b32","slug":"ds_swizzle_b32","records":1,"summary":"Dword swizzle, no data is written to LDS memory.","page":"https://instructionsets.com/amdgpu/ds_swizzle_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_swizzle_b32.json"},{"mnemonic":"ds_wrap_rtn_b32","slug":"ds_wrap_rtn_b32","records":1,"summary":"Given a minuend from a location in data share and a subtrahend from a vector register, subtract the two values iff the result is nonnegative…","page":"https://instructionsets.com/amdgpu/ds_wrap_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_wrap_rtn_b32.json"},{"mnemonic":"ds_write2_b32","slug":"ds_write2_b32","records":1,"summary":"Store 32 bits of data from one vector input register and then 32 bits of data from a second vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write2_b32.json","aliases":["ds_store_2addr_b32"]},{"mnemonic":"ds_write2_b64","slug":"ds_write2_b64","records":1,"summary":"Store 64 bits of data from one vector input register and then 64 bits of data from a second vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write2_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write2_b64.json","aliases":["ds_store_2addr_b64"]},{"mnemonic":"ds_write2st64_b32","slug":"ds_write2st64_b32","records":1,"summary":"Store 32 bits of data from one vector input register and then 32 bits of data from a second vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write2st64_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write2st64_b32.json","aliases":["ds_store_2addr_stride64_b32"]},{"mnemonic":"ds_write2st64_b64","slug":"ds_write2st64_b64","records":1,"summary":"Store 64 bits of data from one vector input register and then 64 bits of data from a second vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write2st64_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write2st64_b64.json","aliases":["ds_store_2addr_stride64_b64"]},{"mnemonic":"ds_write_addtid_b32","slug":"ds_write_addtid_b32","records":1,"summary":"Store 32 bits of data from a vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write_addtid_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_addtid_b32.json","aliases":["ds_store_addtid_b32"]},{"mnemonic":"ds_write_b128","slug":"ds_write_b128","records":1,"summary":"Store 128 bits of data from a vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write_b128/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_b128.json","aliases":["ds_store_b128"]},{"mnemonic":"ds_write_b16","slug":"ds_write_b16","records":1,"summary":"Store 16 bits of data from a vector register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write_b16/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_b16.json","aliases":["ds_store_b16"]},{"mnemonic":"ds_write_b16_d16_hi","slug":"ds_write_b16_d16_hi","records":1,"summary":"Store 16 bits of data from the high bits of a vector register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write_b16_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_b16_d16_hi.json","aliases":["ds_store_b16_d16_hi"]},{"mnemonic":"ds_write_b32","slug":"ds_write_b32","records":1,"summary":"Write one 32-bit value per lane to the Local Data Share (LDS).","page":"https://instructionsets.com/amdgpu/ds_write_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_b32.json","aliases":["ds_store_b32"]},{"mnemonic":"ds_write_b64","slug":"ds_write_b64","records":1,"summary":"Store 64 bits of data from a vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_b64.json","aliases":["ds_store_b64"]},{"mnemonic":"ds_write_b8","slug":"ds_write_b8","records":1,"summary":"Store 8 bits of data from a vector register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write_b8/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_b8.json","aliases":["ds_store_b8"]},{"mnemonic":"ds_write_b8_d16_hi","slug":"ds_write_b8_d16_hi","records":1,"summary":"Store 8 bits of data from the high bits of a vector register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write_b8_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_b8_d16_hi.json","aliases":["ds_store_b8_d16_hi"]},{"mnemonic":"ds_write_b96","slug":"ds_write_b96","records":1,"summary":"Store 96 bits of data from a vector input register into a data share.","page":"https://instructionsets.com/amdgpu/ds_write_b96/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_b96.json","aliases":["ds_store_b96"]},{"mnemonic":"ds_write_src2_b32","slug":"ds_write_src2_b32","records":1,"summary":"AMDGPU DS vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_write_src2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_src2_b32.json"},{"mnemonic":"ds_write_src2_b64","slug":"ds_write_src2_b64","records":1,"summary":"AMDGPU DS vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_write_src2_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_write_src2_b64.json"},{"mnemonic":"ds_wrxchg2_rtn_b32","slug":"ds_wrxchg2_rtn_b32","records":1,"summary":"Swap two unsigned 32-bit integer values in the data registers with two locations in a data share.","page":"https://instructionsets.com/amdgpu/ds_wrxchg2_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_wrxchg2_rtn_b32.json","aliases":["ds_storexchg_2addr_rtn_b32"]},{"mnemonic":"ds_wrxchg2_rtn_b64","slug":"ds_wrxchg2_rtn_b64","records":1,"summary":"Swap two unsigned 64-bit integer values in the data registers with two locations in a data share.","page":"https://instructionsets.com/amdgpu/ds_wrxchg2_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_wrxchg2_rtn_b64.json","aliases":["ds_storexchg_2addr_rtn_b64"]},{"mnemonic":"ds_wrxchg2st64_rtn_b32","slug":"ds_wrxchg2st64_rtn_b32","records":1,"summary":"Swap two unsigned 32-bit integer values in the data registers with two locations in a data share.","page":"https://instructionsets.com/amdgpu/ds_wrxchg2st64_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_wrxchg2st64_rtn_b32.json","aliases":["ds_storexchg_2addr_stride64_rtn_b32"]},{"mnemonic":"ds_wrxchg2st64_rtn_b64","slug":"ds_wrxchg2st64_rtn_b64","records":1,"summary":"Swap two unsigned 64-bit integer values in the data registers with two locations in a data share.","page":"https://instructionsets.com/amdgpu/ds_wrxchg2st64_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_wrxchg2st64_rtn_b64.json","aliases":["ds_storexchg_2addr_stride64_rtn_b64"]},{"mnemonic":"ds_wrxchg_rtn_b32","slug":"ds_wrxchg_rtn_b32","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_wrxchg_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_wrxchg_rtn_b32.json","aliases":["ds_storexchg_rtn_b32"]},{"mnemonic":"ds_wrxchg_rtn_b64","slug":"ds_wrxchg_rtn_b64","records":1,"summary":"Swap an unsigned 64-bit integer value in the data register with a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_wrxchg_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_wrxchg_rtn_b64.json","aliases":["ds_storexchg_rtn_b64"]},{"mnemonic":"ds_xor_b32","slug":"ds_xor_b32","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_xor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_xor_b32.json"},{"mnemonic":"ds_xor_b64","slug":"ds_xor_b64","records":1,"summary":"Calculate bitwise XOR given two unsigned 64-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_xor_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_xor_b64.json"},{"mnemonic":"ds_xor_rtn_b32","slug":"ds_xor_rtn_b32","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_xor_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_xor_rtn_b32.json"},{"mnemonic":"ds_xor_rtn_b64","slug":"ds_xor_rtn_b64","records":1,"summary":"Calculate bitwise XOR given two unsigned 64-bit integer values stored in the data register and a location in a data share.","page":"https://instructionsets.com/amdgpu/ds_xor_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_xor_rtn_b64.json"},{"mnemonic":"ds_xor_src2_b32","slug":"ds_xor_src2_b32","records":1,"summary":"AMDGPU DS vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_xor_src2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/ds_xor_src2_b32.json"},{"mnemonic":"ds_xor_src2_b64","slug":"ds_xor_src2_b64","records":1,"summary":"AMDGPU DS vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/ds_xor_src2_b64/","api":"https://instructionsets.com/api/v1/amdgpu/ds_xor_src2_b64.json"},{"mnemonic":"exp","slug":"exp","records":1,"summary":"Export graphics data to the next stage of the render pipeline. The target and up to four channels of data are specified as operands.","page":"https://instructionsets.com/amdgpu/exp/","api":"https://instructionsets.com/api/v1/amdgpu/exp.json","aliases":["export"]},{"mnemonic":"export","slug":"export","records":1,"summary":"Export graphics data to the next stage of the render pipeline. The target and up to four channels of data are specified as operands.","page":"https://instructionsets.com/amdgpu/export/","api":"https://instructionsets.com/api/v1/amdgpu/export.json","aliases":["exp"]},{"mnemonic":"flat_atomic_add","slug":"flat_atomic_add","records":1,"summary":"Add two unsigned 32-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_add/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_add.json","aliases":["flat_atomic_add_u32"]},{"mnemonic":"flat_atomic_add_f32","slug":"flat_atomic_add_f32","records":1,"summary":"Add a single-precision float value in the data register to a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_add_f32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_add_f32.json"},{"mnemonic":"flat_atomic_add_f64","slug":"flat_atomic_add_f64","records":1,"summary":"Add a double-precision float value in the data register to a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_add_f64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_add_f64.json"},{"mnemonic":"flat_atomic_add_u32","slug":"flat_atomic_add_u32","records":1,"summary":"Add two unsigned 32-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_add_u32.json","aliases":["flat_atomic_add"]},{"mnemonic":"flat_atomic_add_u64","slug":"flat_atomic_add_u64","records":1,"summary":"Add two unsigned 64-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_add_u64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_add_u64.json","aliases":["flat_atomic_add_x2"]},{"mnemonic":"flat_atomic_add_x2","slug":"flat_atomic_add_x2","records":1,"summary":"Add two unsigned 64-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_add_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_add_x2.json","aliases":["flat_atomic_add_u64"]},{"mnemonic":"flat_atomic_and","slug":"flat_atomic_and","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_and/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_and.json","aliases":["flat_atomic_and_b32"]},{"mnemonic":"flat_atomic_and_b32","slug":"flat_atomic_and_b32","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_and_b32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_and_b32.json","aliases":["flat_atomic_and"]},{"mnemonic":"flat_atomic_and_b64","slug":"flat_atomic_and_b64","records":1,"summary":"Calculate bitwise AND given two unsigned 64-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_and_b64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_and_b64.json","aliases":["flat_atomic_and_x2"]},{"mnemonic":"flat_atomic_and_x2","slug":"flat_atomic_and_x2","records":1,"summary":"Calculate bitwise AND given two unsigned 64-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_and_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_and_x2.json","aliases":["flat_atomic_and_b64"]},{"mnemonic":"flat_atomic_cmpswap","slug":"flat_atomic_cmpswap","records":1,"summary":"Compare two unsigned 32-bit integer values stored in the data comparison register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_cmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_cmpswap.json","aliases":["flat_atomic_cmpswap_b32"]},{"mnemonic":"flat_atomic_cmpswap_b32","slug":"flat_atomic_cmpswap_b32","records":1,"summary":"Compare two unsigned 32-bit integer values stored in the data comparison register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_cmpswap_b32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_cmpswap_b32.json","aliases":["flat_atomic_cmpswap"]},{"mnemonic":"flat_atomic_cmpswap_b64","slug":"flat_atomic_cmpswap_b64","records":1,"summary":"Compare two unsigned 64-bit integer values stored in the data comparison register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_cmpswap_b64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_cmpswap_b64.json","aliases":["flat_atomic_cmpswap_x2"]},{"mnemonic":"flat_atomic_cmpswap_f32","slug":"flat_atomic_cmpswap_f32","records":1,"summary":"Compare two single-precision float values stored in the data comparison register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_cmpswap_f32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_cmpswap_f32.json","aliases":["flat_atomic_fcmpswap"]},{"mnemonic":"flat_atomic_cmpswap_x2","slug":"flat_atomic_cmpswap_x2","records":1,"summary":"Compare two unsigned 64-bit integer values stored in the data comparison register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_cmpswap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_cmpswap_x2.json","aliases":["flat_atomic_cmpswap_b64"]},{"mnemonic":"flat_atomic_cond_sub_u32","slug":"flat_atomic_cond_sub_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value in the data register from a location in the flat aperture only if the memory value is greater than or equal…","page":"https://instructionsets.com/amdgpu/flat_atomic_cond_sub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_cond_sub_u32.json"},{"mnemonic":"flat_atomic_csub_u32","slug":"flat_atomic_csub_u32","records":1,"summary":"AMDGPU FLAT vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/flat_atomic_csub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_csub_u32.json","aliases":["flat_atomic_sub_clamp_u32"]},{"mnemonic":"flat_atomic_dec","slug":"flat_atomic_dec","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in the flat aperture with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/flat_atomic_dec/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_dec.json","aliases":["flat_atomic_dec_u32"]},{"mnemonic":"flat_atomic_dec_u32","slug":"flat_atomic_dec_u32","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in the flat aperture with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/flat_atomic_dec_u32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_dec_u32.json","aliases":["flat_atomic_dec"]},{"mnemonic":"flat_atomic_dec_u64","slug":"flat_atomic_dec_u64","records":1,"summary":"Decrement an unsigned 64-bit integer value from a location in the flat aperture with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/flat_atomic_dec_u64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_dec_u64.json","aliases":["flat_atomic_dec_x2"]},{"mnemonic":"flat_atomic_dec_x2","slug":"flat_atomic_dec_x2","records":1,"summary":"Decrement an unsigned 64-bit integer value from a location in the flat aperture with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/flat_atomic_dec_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_dec_x2.json","aliases":["flat_atomic_dec_u64"]},{"mnemonic":"flat_atomic_fcmpswap","slug":"flat_atomic_fcmpswap","records":1,"summary":"Compare two single-precision float values stored in the data comparison register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_fcmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_fcmpswap.json","aliases":["flat_atomic_cmpswap_f32"]},{"mnemonic":"flat_atomic_fcmpswap_x2","slug":"flat_atomic_fcmpswap_x2","records":1,"summary":"Compare two double-precision float values stored in the data comparison register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_fcmpswap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_fcmpswap_x2.json"},{"mnemonic":"flat_atomic_fmax","slug":"flat_atomic_fmax","records":1,"summary":"Select the maximum of two single-precision float inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_fmax/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_fmax.json","aliases":["flat_atomic_max_f32","flat_atomic_max_num_f32"]},{"mnemonic":"flat_atomic_fmax_x2","slug":"flat_atomic_fmax_x2","records":1,"summary":"Select the maximum of two double-precision float inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_fmax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_fmax_x2.json"},{"mnemonic":"flat_atomic_fmin","slug":"flat_atomic_fmin","records":1,"summary":"Select the minimum of two single-precision float inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_fmin/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_fmin.json","aliases":["flat_atomic_min_f32","flat_atomic_min_num_f32"]},{"mnemonic":"flat_atomic_fmin_x2","slug":"flat_atomic_fmin_x2","records":1,"summary":"Select the minimum of two double-precision float inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_fmin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_fmin_x2.json"},{"mnemonic":"flat_atomic_inc","slug":"flat_atomic_inc","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in the flat aperture with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/flat_atomic_inc/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_inc.json","aliases":["flat_atomic_inc_u32"]},{"mnemonic":"flat_atomic_inc_u32","slug":"flat_atomic_inc_u32","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in the flat aperture with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/flat_atomic_inc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_inc_u32.json","aliases":["flat_atomic_inc"]},{"mnemonic":"flat_atomic_inc_u64","slug":"flat_atomic_inc_u64","records":1,"summary":"Increment an unsigned 64-bit integer value from a location in the flat aperture with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/flat_atomic_inc_u64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_inc_u64.json","aliases":["flat_atomic_inc_x2"]},{"mnemonic":"flat_atomic_inc_x2","slug":"flat_atomic_inc_x2","records":1,"summary":"Increment an unsigned 64-bit integer value from a location in the flat aperture with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/flat_atomic_inc_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_inc_x2.json","aliases":["flat_atomic_inc_u64"]},{"mnemonic":"flat_atomic_max_f32","slug":"flat_atomic_max_f32","records":1,"summary":"Select the maximum of two single-precision float inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_max_f32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_max_f32.json","aliases":["flat_atomic_fmax","flat_atomic_max_num_f32"]},{"mnemonic":"flat_atomic_max_f64","slug":"flat_atomic_max_f64","records":1,"summary":"Select the maximum of two double-precision float inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_max_f64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_max_f64.json"},{"mnemonic":"flat_atomic_max_i32","slug":"flat_atomic_max_i32","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_max_i32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_max_i32.json","aliases":["flat_atomic_smax"]},{"mnemonic":"flat_atomic_max_i64","slug":"flat_atomic_max_i64","records":1,"summary":"Select the maximum of two signed 64-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_max_i64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_max_i64.json","aliases":["flat_atomic_smax_x2"]},{"mnemonic":"flat_atomic_max_num_f64","slug":"flat_atomic_max_num_f64","records":1,"summary":"Select the IEEE maximumNumber() of two double-precision float inputs, given two values stored in the data register and a location in the flat…","page":"https://instructionsets.com/amdgpu/flat_atomic_max_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_max_num_f64.json"},{"mnemonic":"flat_atomic_max_u32","slug":"flat_atomic_max_u32","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_max_u32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_max_u32.json","aliases":["flat_atomic_umax"]},{"mnemonic":"flat_atomic_max_u64","slug":"flat_atomic_max_u64","records":1,"summary":"Select the maximum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_max_u64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_max_u64.json","aliases":["flat_atomic_umax_x2"]},{"mnemonic":"flat_atomic_min_f32","slug":"flat_atomic_min_f32","records":1,"summary":"Select the minimum of two single-precision float inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_min_f32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_min_f32.json","aliases":["flat_atomic_fmin","flat_atomic_min_num_f32"]},{"mnemonic":"flat_atomic_min_f64","slug":"flat_atomic_min_f64","records":1,"summary":"Select the minimum of two double-precision float inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_min_f64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_min_f64.json"},{"mnemonic":"flat_atomic_min_i32","slug":"flat_atomic_min_i32","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_min_i32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_min_i32.json","aliases":["flat_atomic_smin"]},{"mnemonic":"flat_atomic_min_i64","slug":"flat_atomic_min_i64","records":1,"summary":"Select the minimum of two signed 64-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_min_i64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_min_i64.json","aliases":["flat_atomic_smin_x2"]},{"mnemonic":"flat_atomic_min_num_f64","slug":"flat_atomic_min_num_f64","records":1,"summary":"Select the IEEE minimumNumber() of two double-precision float inputs, given two values stored in the data register and a location in the flat…","page":"https://instructionsets.com/amdgpu/flat_atomic_min_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_min_num_f64.json"},{"mnemonic":"flat_atomic_min_u32","slug":"flat_atomic_min_u32","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_min_u32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_min_u32.json","aliases":["flat_atomic_umin"]},{"mnemonic":"flat_atomic_min_u64","slug":"flat_atomic_min_u64","records":1,"summary":"Select the minimum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_min_u64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_min_u64.json","aliases":["flat_atomic_umin_x2"]},{"mnemonic":"flat_atomic_or","slug":"flat_atomic_or","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_or/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_or.json","aliases":["flat_atomic_or_b32"]},{"mnemonic":"flat_atomic_or_b32","slug":"flat_atomic_or_b32","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_or_b32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_or_b32.json","aliases":["flat_atomic_or"]},{"mnemonic":"flat_atomic_or_b64","slug":"flat_atomic_or_b64","records":1,"summary":"Calculate bitwise OR given two unsigned 64-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_or_b64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_or_b64.json","aliases":["flat_atomic_or_x2"]},{"mnemonic":"flat_atomic_or_x2","slug":"flat_atomic_or_x2","records":1,"summary":"Calculate bitwise OR given two unsigned 64-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_or_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_or_x2.json","aliases":["flat_atomic_or_b64"]},{"mnemonic":"flat_atomic_pk_add_bf16","slug":"flat_atomic_pk_add_bf16","records":1,"summary":"Add a packed 2-component BF16 float value in the data register to a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_pk_add_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_pk_add_bf16.json"},{"mnemonic":"flat_atomic_pk_add_f16","slug":"flat_atomic_pk_add_f16","records":1,"summary":"Add a packed 2-component half-precision float value in the data register to a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_pk_add_f16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_pk_add_f16.json"},{"mnemonic":"flat_atomic_smax","slug":"flat_atomic_smax","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_smax/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_smax.json","aliases":["flat_atomic_max_i32"]},{"mnemonic":"flat_atomic_smax_x2","slug":"flat_atomic_smax_x2","records":1,"summary":"Select the maximum of two signed 64-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_smax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_smax_x2.json","aliases":["flat_atomic_max_i64"]},{"mnemonic":"flat_atomic_smin","slug":"flat_atomic_smin","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_smin/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_smin.json","aliases":["flat_atomic_min_i32"]},{"mnemonic":"flat_atomic_smin_x2","slug":"flat_atomic_smin_x2","records":1,"summary":"Select the minimum of two signed 64-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_smin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_smin_x2.json","aliases":["flat_atomic_min_i64"]},{"mnemonic":"flat_atomic_sub","slug":"flat_atomic_sub","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_sub/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_sub.json","aliases":["flat_atomic_sub_u32"]},{"mnemonic":"flat_atomic_sub_u32","slug":"flat_atomic_sub_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_sub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_sub_u32.json","aliases":["flat_atomic_sub"]},{"mnemonic":"flat_atomic_sub_u64","slug":"flat_atomic_sub_u64","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in the data register from a value stored in a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_sub_u64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_sub_u64.json","aliases":["flat_atomic_sub_x2"]},{"mnemonic":"flat_atomic_sub_x2","slug":"flat_atomic_sub_x2","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in the data register from a value stored in a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_sub_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_sub_x2.json","aliases":["flat_atomic_sub_u64"]},{"mnemonic":"flat_atomic_swap","slug":"flat_atomic_swap","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_swap/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_swap.json","aliases":["flat_atomic_swap_b32"]},{"mnemonic":"flat_atomic_swap_b32","slug":"flat_atomic_swap_b32","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_swap_b32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_swap_b32.json","aliases":["flat_atomic_swap"]},{"mnemonic":"flat_atomic_swap_b64","slug":"flat_atomic_swap_b64","records":1,"summary":"Swap an unsigned 64-bit integer value in the data register with a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_swap_b64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_swap_b64.json","aliases":["flat_atomic_swap_x2"]},{"mnemonic":"flat_atomic_swap_x2","slug":"flat_atomic_swap_x2","records":1,"summary":"Swap an unsigned 64-bit integer value in the data register with a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_swap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_swap_x2.json","aliases":["flat_atomic_swap_b64"]},{"mnemonic":"flat_atomic_umax","slug":"flat_atomic_umax","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_umax/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_umax.json","aliases":["flat_atomic_max_u32"]},{"mnemonic":"flat_atomic_umax_x2","slug":"flat_atomic_umax_x2","records":1,"summary":"Select the maximum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_umax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_umax_x2.json","aliases":["flat_atomic_max_u64"]},{"mnemonic":"flat_atomic_umin","slug":"flat_atomic_umin","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_umin/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_umin.json","aliases":["flat_atomic_min_u32"]},{"mnemonic":"flat_atomic_umin_x2","slug":"flat_atomic_umin_x2","records":1,"summary":"Select the minimum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_umin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_umin_x2.json","aliases":["flat_atomic_min_u64"]},{"mnemonic":"flat_atomic_xor","slug":"flat_atomic_xor","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_xor/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_xor.json","aliases":["flat_atomic_xor_b32"]},{"mnemonic":"flat_atomic_xor_b32","slug":"flat_atomic_xor_b32","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_xor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_xor_b32.json","aliases":["flat_atomic_xor"]},{"mnemonic":"flat_atomic_xor_b64","slug":"flat_atomic_xor_b64","records":1,"summary":"Calculate bitwise XOR given two unsigned 64-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_xor_b64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_xor_b64.json","aliases":["flat_atomic_xor_x2"]},{"mnemonic":"flat_atomic_xor_x2","slug":"flat_atomic_xor_x2","records":1,"summary":"Calculate bitwise XOR given two unsigned 64-bit integer values stored in the data register and a location in the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_atomic_xor_x2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_atomic_xor_x2.json","aliases":["flat_atomic_xor_b64"]},{"mnemonic":"flat_load_b128","slug":"flat_load_b128","records":1,"summary":"Load 128 bits of data from the flat aperture into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_b128/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_b128.json","aliases":["flat_load_dwordx4"]},{"mnemonic":"flat_load_b32","slug":"flat_load_b32","records":1,"summary":"Load 32 bits of data from the flat aperture into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_b32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_b32.json","aliases":["flat_load_dword"]},{"mnemonic":"flat_load_b64","slug":"flat_load_b64","records":1,"summary":"Load 64 bits of data from the flat aperture into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_b64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_b64.json","aliases":["flat_load_dwordx2"]},{"mnemonic":"flat_load_b96","slug":"flat_load_b96","records":1,"summary":"Load 96 bits of data from the flat aperture into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_b96/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_b96.json","aliases":["flat_load_dwordx3"]},{"mnemonic":"flat_load_d16_b16","slug":"flat_load_d16_b16","records":1,"summary":"Load 16 bits of unsigned data from the flat aperture and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_d16_b16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_d16_b16.json","aliases":["flat_load_short_d16"]},{"mnemonic":"flat_load_d16_hi_b16","slug":"flat_load_d16_hi_b16","records":1,"summary":"Load 16 bits of unsigned data from the flat aperture and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_d16_hi_b16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_d16_hi_b16.json","aliases":["flat_load_short_d16_hi"]},{"mnemonic":"flat_load_d16_hi_i8","slug":"flat_load_d16_hi_i8","records":1,"summary":"Load 8 bits of signed data from the flat aperture, sign extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_d16_hi_i8/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_d16_hi_i8.json","aliases":["flat_load_sbyte_d16_hi"]},{"mnemonic":"flat_load_d16_hi_u8","slug":"flat_load_d16_hi_u8","records":1,"summary":"Load 8 bits of unsigned data from the flat aperture, zero extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_d16_hi_u8/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_d16_hi_u8.json","aliases":["flat_load_ubyte_d16_hi"]},{"mnemonic":"flat_load_d16_i8","slug":"flat_load_d16_i8","records":1,"summary":"Load 8 bits of signed data from the flat aperture, sign extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_d16_i8/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_d16_i8.json","aliases":["flat_load_sbyte_d16"]},{"mnemonic":"flat_load_d16_u8","slug":"flat_load_d16_u8","records":1,"summary":"Load 8 bits of unsigned data from the flat aperture, zero extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_d16_u8/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_d16_u8.json","aliases":["flat_load_ubyte_d16"]},{"mnemonic":"flat_load_dword","slug":"flat_load_dword","records":1,"summary":"Load one 32-bit dword per lane through the flat (generic) address space, resolved to global/scratch/LDS at runtime.","page":"https://instructionsets.com/amdgpu/flat_load_dword/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_dword.json","aliases":["flat_load_b32"]},{"mnemonic":"flat_load_dwordx2","slug":"flat_load_dwordx2","records":1,"summary":"Load 64 bits of data from the flat aperture into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_dwordx2.json","aliases":["flat_load_b64"]},{"mnemonic":"flat_load_dwordx3","slug":"flat_load_dwordx3","records":1,"summary":"Load 96 bits of data from the flat aperture into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_dwordx3/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_dwordx3.json","aliases":["flat_load_b96"]},{"mnemonic":"flat_load_dwordx4","slug":"flat_load_dwordx4","records":1,"summary":"Load 128 bits of data from the flat aperture into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_dwordx4.json","aliases":["flat_load_b128"]},{"mnemonic":"flat_load_i16","slug":"flat_load_i16","records":1,"summary":"Load 16 bits of signed data from the flat aperture, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_i16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_i16.json","aliases":["flat_load_sshort"]},{"mnemonic":"flat_load_i8","slug":"flat_load_i8","records":1,"summary":"Load 8 bits of signed data from the flat aperture, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_i8/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_i8.json","aliases":["flat_load_sbyte"]},{"mnemonic":"flat_load_monitor_b128","slug":"flat_load_monitor_b128","records":1,"summary":"AMDGPU FLAT vector instruction operating on b128 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/flat_load_monitor_b128/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_monitor_b128.json"},{"mnemonic":"flat_load_monitor_b32","slug":"flat_load_monitor_b32","records":1,"summary":"AMDGPU FLAT vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/flat_load_monitor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_monitor_b32.json"},{"mnemonic":"flat_load_monitor_b64","slug":"flat_load_monitor_b64","records":1,"summary":"AMDGPU FLAT vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/flat_load_monitor_b64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_monitor_b64.json"},{"mnemonic":"flat_load_sbyte","slug":"flat_load_sbyte","records":1,"summary":"Load 8 bits of signed data from the flat aperture, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_sbyte/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_sbyte.json","aliases":["flat_load_i8"]},{"mnemonic":"flat_load_sbyte_d16","slug":"flat_load_sbyte_d16","records":1,"summary":"Load 8 bits of signed data from the flat aperture, sign extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_sbyte_d16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_sbyte_d16.json","aliases":["flat_load_d16_i8"]},{"mnemonic":"flat_load_sbyte_d16_hi","slug":"flat_load_sbyte_d16_hi","records":1,"summary":"Load 8 bits of signed data from the flat aperture, sign extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_sbyte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_sbyte_d16_hi.json","aliases":["flat_load_d16_hi_i8"]},{"mnemonic":"flat_load_short_d16","slug":"flat_load_short_d16","records":1,"summary":"Load 16 bits of unsigned data from the flat aperture and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_short_d16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_short_d16.json","aliases":["flat_load_d16_b16"]},{"mnemonic":"flat_load_short_d16_hi","slug":"flat_load_short_d16_hi","records":1,"summary":"Load 16 bits of unsigned data from the flat aperture and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_short_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_short_d16_hi.json","aliases":["flat_load_d16_hi_b16"]},{"mnemonic":"flat_load_sshort","slug":"flat_load_sshort","records":1,"summary":"Load 16 bits of signed data from the flat aperture, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_sshort/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_sshort.json","aliases":["flat_load_i16"]},{"mnemonic":"flat_load_u16","slug":"flat_load_u16","records":1,"summary":"Load 16 bits of unsigned data from the flat aperture, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_u16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_u16.json","aliases":["flat_load_ushort"]},{"mnemonic":"flat_load_u8","slug":"flat_load_u8","records":1,"summary":"Load 8 bits of unsigned data from the flat aperture, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_u8/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_u8.json","aliases":["flat_load_ubyte"]},{"mnemonic":"flat_load_ubyte","slug":"flat_load_ubyte","records":1,"summary":"Load 8 bits of unsigned data from the flat aperture, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_ubyte/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_ubyte.json","aliases":["flat_load_u8"]},{"mnemonic":"flat_load_ubyte_d16","slug":"flat_load_ubyte_d16","records":1,"summary":"Load 8 bits of unsigned data from the flat aperture, zero extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_ubyte_d16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_ubyte_d16.json","aliases":["flat_load_d16_u8"]},{"mnemonic":"flat_load_ubyte_d16_hi","slug":"flat_load_ubyte_d16_hi","records":1,"summary":"Load 8 bits of unsigned data from the flat aperture, zero extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/flat_load_ubyte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_ubyte_d16_hi.json","aliases":["flat_load_d16_hi_u8"]},{"mnemonic":"flat_load_ushort","slug":"flat_load_ushort","records":1,"summary":"Load 16 bits of unsigned data from the flat aperture, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/flat_load_ushort/","api":"https://instructionsets.com/api/v1/amdgpu/flat_load_ushort.json","aliases":["flat_load_u16"]},{"mnemonic":"flat_prefetch_b8","slug":"flat_prefetch_b8","records":1,"summary":"AMDGPU FLAT vector instruction operating on b8 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/flat_prefetch_b8/","api":"https://instructionsets.com/api/v1/amdgpu/flat_prefetch_b8.json"},{"mnemonic":"flat_store_b128","slug":"flat_store_b128","records":1,"summary":"Store 128 bits of data from vector input registers into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_b128/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_b128.json","aliases":["flat_store_dwordx4"]},{"mnemonic":"flat_store_b16","slug":"flat_store_b16","records":1,"summary":"Store 16 bits of data from a vector register into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_b16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_b16.json","aliases":["flat_store_short"]},{"mnemonic":"flat_store_b32","slug":"flat_store_b32","records":1,"summary":"Store 32 bits of data from vector input registers into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_b32/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_b32.json","aliases":["flat_store_dword"]},{"mnemonic":"flat_store_b64","slug":"flat_store_b64","records":1,"summary":"Store 64 bits of data from vector input registers into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_b64/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_b64.json","aliases":["flat_store_dwordx2"]},{"mnemonic":"flat_store_b8","slug":"flat_store_b8","records":1,"summary":"Store 8 bits of data from a vector register into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_b8/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_b8.json","aliases":["flat_store_byte"]},{"mnemonic":"flat_store_b96","slug":"flat_store_b96","records":1,"summary":"Store 96 bits of data from vector input registers into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_b96/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_b96.json","aliases":["flat_store_dwordx3"]},{"mnemonic":"flat_store_byte","slug":"flat_store_byte","records":1,"summary":"Store 8 bits of data from a vector register into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_byte/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_byte.json","aliases":["flat_store_b8"]},{"mnemonic":"flat_store_byte_d16_hi","slug":"flat_store_byte_d16_hi","records":1,"summary":"Store 8 bits of data from the high 16 bits of a 32-bit vector register into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_byte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_byte_d16_hi.json","aliases":["flat_store_d16_hi_b8"]},{"mnemonic":"flat_store_d16_hi_b16","slug":"flat_store_d16_hi_b16","records":1,"summary":"Store 16 bits of data from the high 16 bits of a 32-bit vector register into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_d16_hi_b16/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_d16_hi_b16.json","aliases":["flat_store_short_d16_hi"]},{"mnemonic":"flat_store_d16_hi_b8","slug":"flat_store_d16_hi_b8","records":1,"summary":"Store 8 bits of data from the high 16 bits of a 32-bit vector register into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_d16_hi_b8/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_d16_hi_b8.json","aliases":["flat_store_byte_d16_hi"]},{"mnemonic":"flat_store_dword","slug":"flat_store_dword","records":1,"summary":"Store 32 bits of data from vector input registers into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_dword/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_dword.json","aliases":["flat_store_b32"]},{"mnemonic":"flat_store_dwordx2","slug":"flat_store_dwordx2","records":1,"summary":"Store 64 bits of data from vector input registers into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_dwordx2.json","aliases":["flat_store_b64"]},{"mnemonic":"flat_store_dwordx3","slug":"flat_store_dwordx3","records":1,"summary":"Store 96 bits of data from vector input registers into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_dwordx3/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_dwordx3.json","aliases":["flat_store_b96"]},{"mnemonic":"flat_store_dwordx4","slug":"flat_store_dwordx4","records":1,"summary":"Store 128 bits of data from vector input registers into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_dwordx4.json","aliases":["flat_store_b128"]},{"mnemonic":"flat_store_short","slug":"flat_store_short","records":1,"summary":"Store 16 bits of data from a vector register into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_short/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_short.json","aliases":["flat_store_b16"]},{"mnemonic":"flat_store_short_d16_hi","slug":"flat_store_short_d16_hi","records":1,"summary":"Store 16 bits of data from the high 16 bits of a 32-bit vector register into the flat aperture.","page":"https://instructionsets.com/amdgpu/flat_store_short_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/flat_store_short_d16_hi.json","aliases":["flat_store_d16_hi_b16"]},{"mnemonic":"global_atomic_add","slug":"global_atomic_add","records":1,"summary":"Atomically add a per-lane value to a global-memory location.","page":"https://instructionsets.com/amdgpu/global_atomic_add/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_add.json","aliases":["global_atomic_add_u32"]},{"mnemonic":"global_atomic_add_f32","slug":"global_atomic_add_f32","records":1,"summary":"Add two single-precision float values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_add_f32/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_add_f32.json"},{"mnemonic":"global_atomic_add_f64","slug":"global_atomic_add_f64","records":1,"summary":"Add a double-precision float value in the data register to a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_add_f64/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_add_f64.json"},{"mnemonic":"global_atomic_add_x2","slug":"global_atomic_add_x2","records":1,"summary":"Add two unsigned 64-bit integer values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_add_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_add_x2.json","aliases":["global_atomic_add_u64"]},{"mnemonic":"global_atomic_and","slug":"global_atomic_and","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_and/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_and.json","aliases":["global_atomic_and_b32"]},{"mnemonic":"global_atomic_and_x2","slug":"global_atomic_and_x2","records":1,"summary":"Calculate bitwise AND given two unsigned 64-bit integer values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_and_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_and_x2.json","aliases":["global_atomic_and_b64"]},{"mnemonic":"global_atomic_cmpswap","slug":"global_atomic_cmpswap","records":1,"summary":"Compare two unsigned 32-bit integer values stored in the data comparison register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_cmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_cmpswap.json","aliases":["global_atomic_cmpswap_b32"]},{"mnemonic":"global_atomic_cmpswap_f32","slug":"global_atomic_cmpswap_f32","records":1,"summary":"Compare two single-precision float values stored in the data comparison register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_cmpswap_f32/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_cmpswap_f32.json","aliases":["global_atomic_fcmpswap"]},{"mnemonic":"global_atomic_cmpswap_x2","slug":"global_atomic_cmpswap_x2","records":1,"summary":"Compare two unsigned 64-bit integer values stored in the data comparison register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_cmpswap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_cmpswap_x2.json","aliases":["global_atomic_cmpswap_b64"]},{"mnemonic":"global_atomic_cond_sub_u32","slug":"global_atomic_cond_sub_u32","records":1,"summary":"Subtract an unsigned 32-bit integer value in the data register from a location in the global aperture only if the memory value is greater than or…","page":"https://instructionsets.com/amdgpu/global_atomic_cond_sub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_cond_sub_u32.json"},{"mnemonic":"global_atomic_csub","slug":"global_atomic_csub","records":1,"summary":"Subtract an unsigned 32-bit integer location in the global aperture from a value in the data register and clamp the result to zero.","page":"https://instructionsets.com/amdgpu/global_atomic_csub/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_csub.json","aliases":["global_atomic_csub_u32","global_atomic_sub_clamp_u32"]},{"mnemonic":"global_atomic_dec","slug":"global_atomic_dec","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in the global aperture with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/global_atomic_dec/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_dec.json","aliases":["global_atomic_dec_u32"]},{"mnemonic":"global_atomic_dec_x2","slug":"global_atomic_dec_x2","records":1,"summary":"Decrement an unsigned 64-bit integer value from a location in the global aperture with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/global_atomic_dec_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_dec_x2.json","aliases":["global_atomic_dec_u64"]},{"mnemonic":"global_atomic_fcmpswap","slug":"global_atomic_fcmpswap","records":1,"summary":"Compare two single-precision float values stored in the data comparison register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_fcmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_fcmpswap.json","aliases":["global_atomic_cmpswap_f32"]},{"mnemonic":"global_atomic_fcmpswap_x2","slug":"global_atomic_fcmpswap_x2","records":1,"summary":"Compare two double-precision float values stored in the data comparison register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_fcmpswap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_fcmpswap_x2.json"},{"mnemonic":"global_atomic_fmax","slug":"global_atomic_fmax","records":1,"summary":"Select the maximum of two single-precision float inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_fmax/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_fmax.json","aliases":["global_atomic_max_f32","global_atomic_max_num_f32"]},{"mnemonic":"global_atomic_fmax_x2","slug":"global_atomic_fmax_x2","records":1,"summary":"Select the maximum of two double-precision float inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_fmax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_fmax_x2.json"},{"mnemonic":"global_atomic_fmin","slug":"global_atomic_fmin","records":1,"summary":"Select the minimum of two single-precision float inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_fmin/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_fmin.json","aliases":["global_atomic_min_f32","global_atomic_min_num_f32"]},{"mnemonic":"global_atomic_fmin_x2","slug":"global_atomic_fmin_x2","records":1,"summary":"Select the minimum of two double-precision float inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_fmin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_fmin_x2.json"},{"mnemonic":"global_atomic_inc","slug":"global_atomic_inc","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in the global aperture with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/global_atomic_inc/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_inc.json","aliases":["global_atomic_inc_u32"]},{"mnemonic":"global_atomic_inc_x2","slug":"global_atomic_inc_x2","records":1,"summary":"Increment an unsigned 64-bit integer value from a location in the global aperture with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/global_atomic_inc_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_inc_x2.json","aliases":["global_atomic_inc_u64"]},{"mnemonic":"global_atomic_max_f32","slug":"global_atomic_max_f32","records":1,"summary":"Select the maximum of two single-precision float inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_max_f32/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_max_f32.json","aliases":["global_atomic_fmax","global_atomic_max_num_f32"]},{"mnemonic":"global_atomic_max_f64","slug":"global_atomic_max_f64","records":1,"summary":"Select the maximum of two double-precision float inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_max_f64/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_max_f64.json"},{"mnemonic":"global_atomic_max_num_f32","slug":"global_atomic_max_num_f32","records":1,"summary":"Select the IEEE maximumNumber() of two single-precision float inputs, given two values stored in the data register and a location in the global…","page":"https://instructionsets.com/amdgpu/global_atomic_max_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_max_num_f32.json","aliases":["global_atomic_fmax","global_atomic_max_f32"]},{"mnemonic":"global_atomic_max_num_f64","slug":"global_atomic_max_num_f64","records":1,"summary":"Select the IEEE maximumNumber() of two double-precision float inputs, given two values stored in the data register and a location in the global…","page":"https://instructionsets.com/amdgpu/global_atomic_max_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_max_num_f64.json"},{"mnemonic":"global_atomic_min_f32","slug":"global_atomic_min_f32","records":1,"summary":"Select the minimum of two single-precision float inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_min_f32/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_min_f32.json","aliases":["global_atomic_fmin","global_atomic_min_num_f32"]},{"mnemonic":"global_atomic_min_f64","slug":"global_atomic_min_f64","records":1,"summary":"Select the minimum of two double-precision float inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_min_f64/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_min_f64.json"},{"mnemonic":"global_atomic_min_num_f32","slug":"global_atomic_min_num_f32","records":1,"summary":"Select the IEEE minimumNumber() of two single-precision float inputs, given two values stored in the data register and a location in the global…","page":"https://instructionsets.com/amdgpu/global_atomic_min_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_min_num_f32.json","aliases":["global_atomic_fmin","global_atomic_min_f32"]},{"mnemonic":"global_atomic_min_num_f64","slug":"global_atomic_min_num_f64","records":1,"summary":"Select the IEEE minimumNumber() of two double-precision float inputs, given two values stored in the data register and a location in the global…","page":"https://instructionsets.com/amdgpu/global_atomic_min_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_min_num_f64.json"},{"mnemonic":"global_atomic_or","slug":"global_atomic_or","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_or/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_or.json","aliases":["global_atomic_or_b32"]},{"mnemonic":"global_atomic_or_x2","slug":"global_atomic_or_x2","records":1,"summary":"Calculate bitwise OR given two unsigned 64-bit integer values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_or_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_or_x2.json","aliases":["global_atomic_or_b64"]},{"mnemonic":"global_atomic_ordered_add_b64","slug":"global_atomic_ordered_add_b64","records":1,"summary":"Given an (ID, value) pair in memory, increment the value by a given amount if the ID matches an ID provided by the shader.","page":"https://instructionsets.com/amdgpu/global_atomic_ordered_add_b64/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_ordered_add_b64.json"},{"mnemonic":"global_atomic_pk_add_bf16","slug":"global_atomic_pk_add_bf16","records":1,"summary":"Add a packed 2-component BF16 float value in the data register to a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_pk_add_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_pk_add_bf16.json"},{"mnemonic":"global_atomic_pk_add_f16","slug":"global_atomic_pk_add_f16","records":1,"summary":"Add a packed 2-component half-precision float value from the data register to a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_pk_add_f16/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_pk_add_f16.json"},{"mnemonic":"global_atomic_smax","slug":"global_atomic_smax","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_smax/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_smax.json","aliases":["global_atomic_max_i32"]},{"mnemonic":"global_atomic_smax_x2","slug":"global_atomic_smax_x2","records":1,"summary":"Select the maximum of two signed 64-bit integer inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_smax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_smax_x2.json","aliases":["global_atomic_max_i64"]},{"mnemonic":"global_atomic_smin","slug":"global_atomic_smin","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_smin/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_smin.json","aliases":["global_atomic_min_i32"]},{"mnemonic":"global_atomic_smin_x2","slug":"global_atomic_smin_x2","records":1,"summary":"Select the minimum of two signed 64-bit integer inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_smin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_smin_x2.json","aliases":["global_atomic_min_i64"]},{"mnemonic":"global_atomic_sub","slug":"global_atomic_sub","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_sub/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_sub.json","aliases":["global_atomic_sub_u32"]},{"mnemonic":"global_atomic_sub_x2","slug":"global_atomic_sub_x2","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in the data register from a value stored in a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_sub_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_sub_x2.json","aliases":["global_atomic_sub_u64"]},{"mnemonic":"global_atomic_swap","slug":"global_atomic_swap","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_swap/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_swap.json","aliases":["global_atomic_swap_b32"]},{"mnemonic":"global_atomic_swap_x2","slug":"global_atomic_swap_x2","records":1,"summary":"Swap an unsigned 64-bit integer value in the data register with a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_swap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_swap_x2.json","aliases":["global_atomic_swap_b64"]},{"mnemonic":"global_atomic_umax","slug":"global_atomic_umax","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_umax/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_umax.json","aliases":["global_atomic_max_u32"]},{"mnemonic":"global_atomic_umax_x2","slug":"global_atomic_umax_x2","records":1,"summary":"Select the maximum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_umax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_umax_x2.json","aliases":["global_atomic_max_u64"]},{"mnemonic":"global_atomic_umin","slug":"global_atomic_umin","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_umin/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_umin.json","aliases":["global_atomic_min_u32"]},{"mnemonic":"global_atomic_umin_x2","slug":"global_atomic_umin_x2","records":1,"summary":"Select the minimum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_umin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_umin_x2.json","aliases":["global_atomic_min_u64"]},{"mnemonic":"global_atomic_xor","slug":"global_atomic_xor","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_xor/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_xor.json","aliases":["global_atomic_xor_b32"]},{"mnemonic":"global_atomic_xor_x2","slug":"global_atomic_xor_x2","records":1,"summary":"Calculate bitwise XOR given two unsigned 64-bit integer values stored in the data register and a location in the global aperture.","page":"https://instructionsets.com/amdgpu/global_atomic_xor_x2/","api":"https://instructionsets.com/api/v1/amdgpu/global_atomic_xor_x2.json","aliases":["global_atomic_xor_b64"]},{"mnemonic":"global_inv","slug":"global_inv","records":1,"summary":"Invalidate cache lines based on the SCOPE field. Increments/decrements LOAD_CNT.","page":"https://instructionsets.com/amdgpu/global_inv/","api":"https://instructionsets.com/api/v1/amdgpu/global_inv.json"},{"mnemonic":"global_load_async_to_lds_b128","slug":"global_load_async_to_lds_b128","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b128 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_async_to_lds_b128/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_async_to_lds_b128.json"},{"mnemonic":"global_load_async_to_lds_b32","slug":"global_load_async_to_lds_b32","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_async_to_lds_b32/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_async_to_lds_b32.json"},{"mnemonic":"global_load_async_to_lds_b64","slug":"global_load_async_to_lds_b64","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_async_to_lds_b64/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_async_to_lds_b64.json"},{"mnemonic":"global_load_async_to_lds_b8","slug":"global_load_async_to_lds_b8","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b8 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_async_to_lds_b8/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_async_to_lds_b8.json"},{"mnemonic":"global_load_block","slug":"global_load_block","records":1,"summary":"Load a block of data from the global aperture.","page":"https://instructionsets.com/amdgpu/global_load_block/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_block.json"},{"mnemonic":"global_load_dword","slug":"global_load_dword","records":1,"summary":"Load one 32-bit dword per lane from the global address space using a 64-bit per-lane address.","page":"https://instructionsets.com/amdgpu/global_load_dword/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_dword.json","aliases":["global_load_b32"]},{"mnemonic":"global_load_dword_addtid","slug":"global_load_dword_addtid","records":1,"summary":"Load 32 bits of data from the global aperture into a vector register.","page":"https://instructionsets.com/amdgpu/global_load_dword_addtid/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_dword_addtid.json","aliases":["global_load_addtid_b32"]},{"mnemonic":"global_load_dwordx2","slug":"global_load_dwordx2","records":1,"summary":"Load 64 bits of data from the global aperture into a vector register.","page":"https://instructionsets.com/amdgpu/global_load_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_dwordx2.json","aliases":["global_load_b64"]},{"mnemonic":"global_load_dwordx3","slug":"global_load_dwordx3","records":1,"summary":"Load 96 bits of data from the global aperture into a vector register.","page":"https://instructionsets.com/amdgpu/global_load_dwordx3/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_dwordx3.json","aliases":["global_load_b96"]},{"mnemonic":"global_load_dwordx4","slug":"global_load_dwordx4","records":1,"summary":"Load 128 bits of data from the global aperture into a vector register.","page":"https://instructionsets.com/amdgpu/global_load_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_dwordx4.json","aliases":["global_load_b128"]},{"mnemonic":"global_load_lds_dword","slug":"global_load_lds_dword","records":1,"summary":"Load 32 bits of untyped data from the global aperture and store the result into a data share.","page":"https://instructionsets.com/amdgpu/global_load_lds_dword/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_lds_dword.json"},{"mnemonic":"global_load_lds_dwordx3","slug":"global_load_lds_dwordx3","records":1,"summary":"Untyped buffer load 3 dwords, store result into data share.","page":"https://instructionsets.com/amdgpu/global_load_lds_dwordx3/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_lds_dwordx3.json"},{"mnemonic":"global_load_lds_dwordx4","slug":"global_load_lds_dwordx4","records":1,"summary":"Untyped buffer load 4 dwords, store result into data share.","page":"https://instructionsets.com/amdgpu/global_load_lds_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_lds_dwordx4.json"},{"mnemonic":"global_load_lds_sbyte","slug":"global_load_lds_sbyte","records":1,"summary":"Load 8 bits of untyped data from the global aperture, sign extend to 32 bits and store the result into a data share.","page":"https://instructionsets.com/amdgpu/global_load_lds_sbyte/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_lds_sbyte.json"},{"mnemonic":"global_load_lds_sshort","slug":"global_load_lds_sshort","records":1,"summary":"Load 16 bits of untyped data from the global aperture, sign extend to 32 bits and store the result into a data share.","page":"https://instructionsets.com/amdgpu/global_load_lds_sshort/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_lds_sshort.json"},{"mnemonic":"global_load_lds_ubyte","slug":"global_load_lds_ubyte","records":1,"summary":"Load 8 bits of untyped data from the global aperture, zero extend to 32 bits and store the result into a data share.","page":"https://instructionsets.com/amdgpu/global_load_lds_ubyte/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_lds_ubyte.json"},{"mnemonic":"global_load_lds_ushort","slug":"global_load_lds_ushort","records":1,"summary":"Load 16 bits of untyped data from the global aperture, zero extend to 32 bits and store the result into a data share.","page":"https://instructionsets.com/amdgpu/global_load_lds_ushort/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_lds_ushort.json"},{"mnemonic":"global_load_monitor_b128","slug":"global_load_monitor_b128","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b128 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_monitor_b128/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_monitor_b128.json"},{"mnemonic":"global_load_monitor_b32","slug":"global_load_monitor_b32","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_monitor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_monitor_b32.json"},{"mnemonic":"global_load_monitor_b64","slug":"global_load_monitor_b64","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_monitor_b64/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_monitor_b64.json"},{"mnemonic":"global_load_sbyte","slug":"global_load_sbyte","records":1,"summary":"Load 8 bits of signed data from the global aperture, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/global_load_sbyte/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_sbyte.json","aliases":["global_load_i8"]},{"mnemonic":"global_load_sbyte_d16","slug":"global_load_sbyte_d16","records":1,"summary":"Load 8 bits of signed data from the global aperture, sign extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/global_load_sbyte_d16/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_sbyte_d16.json","aliases":["global_load_d16_i8"]},{"mnemonic":"global_load_sbyte_d16_hi","slug":"global_load_sbyte_d16_hi","records":1,"summary":"Load 8 bits of signed data from the global aperture, sign extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/global_load_sbyte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_sbyte_d16_hi.json","aliases":["global_load_d16_hi_i8"]},{"mnemonic":"global_load_short_d16","slug":"global_load_short_d16","records":1,"summary":"Load 16 bits of unsigned data from the global aperture and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/global_load_short_d16/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_short_d16.json","aliases":["global_load_d16_b16"]},{"mnemonic":"global_load_short_d16_hi","slug":"global_load_short_d16_hi","records":1,"summary":"Load 16 bits of unsigned data from the global aperture and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/global_load_short_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_short_d16_hi.json","aliases":["global_load_d16_hi_b16"]},{"mnemonic":"global_load_sshort","slug":"global_load_sshort","records":1,"summary":"Load 16 bits of signed data from the global aperture, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/global_load_sshort/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_sshort.json","aliases":["global_load_i16"]},{"mnemonic":"global_load_tr4_b64","slug":"global_load_tr4_b64","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_tr4_b64/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_tr4_b64.json","aliases":["global_load_b64_tr_b4"]},{"mnemonic":"global_load_tr6_b96","slug":"global_load_tr6_b96","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b96 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_tr6_b96/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_tr6_b96.json","aliases":["global_load_b128_tr_b6"]},{"mnemonic":"global_load_tr_b128","slug":"global_load_tr_b128","records":1,"summary":"Load a 16x16 matrix of 16-bit data from the global aperture, transpose data between row-major and column-major order, and store the result into a…","page":"https://instructionsets.com/amdgpu/global_load_tr_b128/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_tr_b128.json","aliases":["global_load_b128_tr_b16","global_load_tr16_b128"]},{"mnemonic":"global_load_tr_b128_w64","slug":"global_load_tr_b128_w64","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b128 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_tr_b128_w64/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_tr_b128_w64.json"},{"mnemonic":"global_load_tr_b64","slug":"global_load_tr_b64","records":1,"summary":"Load a 16x16 matrix of 8-bit data from the global aperture, transpose data between row-major and column-major order, and store the result into a…","page":"https://instructionsets.com/amdgpu/global_load_tr_b64/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_tr_b64.json","aliases":["global_load_b64_tr_b8","global_load_tr8_b64"]},{"mnemonic":"global_load_tr_b64_w64","slug":"global_load_tr_b64_w64","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_load_tr_b64_w64/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_tr_b64_w64.json"},{"mnemonic":"global_load_ubyte","slug":"global_load_ubyte","records":1,"summary":"Load 8 bits of unsigned data from the global aperture, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/global_load_ubyte/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_ubyte.json","aliases":["global_load_u8"]},{"mnemonic":"global_load_ubyte_d16","slug":"global_load_ubyte_d16","records":1,"summary":"Load 8 bits of unsigned data from the global aperture, zero extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/global_load_ubyte_d16/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_ubyte_d16.json","aliases":["global_load_d16_u8"]},{"mnemonic":"global_load_ubyte_d16_hi","slug":"global_load_ubyte_d16_hi","records":1,"summary":"Load 8 bits of unsigned data from the global aperture, zero extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/global_load_ubyte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_ubyte_d16_hi.json","aliases":["global_load_d16_hi_u8"]},{"mnemonic":"global_load_ushort","slug":"global_load_ushort","records":1,"summary":"Load 16 bits of unsigned data from the global aperture, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/global_load_ushort/","api":"https://instructionsets.com/api/v1/amdgpu/global_load_ushort.json","aliases":["global_load_u16"]},{"mnemonic":"global_prefetch_b8","slug":"global_prefetch_b8","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b8 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_prefetch_b8/","api":"https://instructionsets.com/api/v1/amdgpu/global_prefetch_b8.json"},{"mnemonic":"global_store_async_from_lds_b128","slug":"global_store_async_from_lds_b128","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b128 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_store_async_from_lds_b128/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_async_from_lds_b128.json"},{"mnemonic":"global_store_async_from_lds_b32","slug":"global_store_async_from_lds_b32","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_store_async_from_lds_b32/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_async_from_lds_b32.json"},{"mnemonic":"global_store_async_from_lds_b64","slug":"global_store_async_from_lds_b64","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_store_async_from_lds_b64/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_async_from_lds_b64.json"},{"mnemonic":"global_store_async_from_lds_b8","slug":"global_store_async_from_lds_b8","records":1,"summary":"AMDGPU GLOBAL vector instruction operating on b8 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/global_store_async_from_lds_b8/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_async_from_lds_b8.json"},{"mnemonic":"global_store_block","slug":"global_store_block","records":1,"summary":"Store a block of data to the global aperture.","page":"https://instructionsets.com/amdgpu/global_store_block/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_block.json"},{"mnemonic":"global_store_byte","slug":"global_store_byte","records":1,"summary":"Store 8 bits of data from a vector register into the global aperture.","page":"https://instructionsets.com/amdgpu/global_store_byte/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_byte.json","aliases":["global_store_b8"]},{"mnemonic":"global_store_byte_d16_hi","slug":"global_store_byte_d16_hi","records":1,"summary":"Store 8 bits of data from the high 16 bits of a 32-bit vector register into the global aperture.","page":"https://instructionsets.com/amdgpu/global_store_byte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_byte_d16_hi.json","aliases":["global_store_d16_hi_b8"]},{"mnemonic":"global_store_dword","slug":"global_store_dword","records":1,"summary":"Store one 32-bit dword per lane to the global address space using a 64-bit per-lane address.","page":"https://instructionsets.com/amdgpu/global_store_dword/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_dword.json","aliases":["global_store_b32"]},{"mnemonic":"global_store_dword_addtid","slug":"global_store_dword_addtid","records":1,"summary":"Store 32 bits of data from a vector input register into the global aperture.","page":"https://instructionsets.com/amdgpu/global_store_dword_addtid/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_dword_addtid.json","aliases":["global_store_addtid_b32"]},{"mnemonic":"global_store_dwordx2","slug":"global_store_dwordx2","records":1,"summary":"Store 64 bits of data from vector input registers into the global aperture.","page":"https://instructionsets.com/amdgpu/global_store_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_dwordx2.json","aliases":["global_store_b64"]},{"mnemonic":"global_store_dwordx3","slug":"global_store_dwordx3","records":1,"summary":"Store 96 bits of data from vector input registers into the global aperture.","page":"https://instructionsets.com/amdgpu/global_store_dwordx3/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_dwordx3.json","aliases":["global_store_b96"]},{"mnemonic":"global_store_dwordx4","slug":"global_store_dwordx4","records":1,"summary":"Store 128 bits of data from vector input registers into the global aperture.","page":"https://instructionsets.com/amdgpu/global_store_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_dwordx4.json","aliases":["global_store_b128"]},{"mnemonic":"global_store_short","slug":"global_store_short","records":1,"summary":"Store 16 bits of data from a vector register into the global aperture.","page":"https://instructionsets.com/amdgpu/global_store_short/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_short.json","aliases":["global_store_b16"]},{"mnemonic":"global_store_short_d16_hi","slug":"global_store_short_d16_hi","records":1,"summary":"Store 16 bits of data from the high 16 bits of a 32-bit vector register into the global aperture.","page":"https://instructionsets.com/amdgpu/global_store_short_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/global_store_short_d16_hi.json","aliases":["global_store_d16_hi_b16"]},{"mnemonic":"global_wb","slug":"global_wb","records":1,"summary":"Write back dirty cache lines based on the SCOPE field. Increments/decrements STORE_CNT.","page":"https://instructionsets.com/amdgpu/global_wb/","api":"https://instructionsets.com/api/v1/amdgpu/global_wb.json"},{"mnemonic":"global_wbinv","slug":"global_wbinv","records":1,"summary":"Write back and invalidate cache lines based on the SCOPE field. Increments/decrements STORE_CNT.","page":"https://instructionsets.com/amdgpu/global_wbinv/","api":"https://instructionsets.com/api/v1/amdgpu/global_wbinv.json"},{"mnemonic":"image_atomic_add","slug":"image_atomic_add","records":1,"summary":"Add two unsigned 32-bit integer values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_add/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_add.json","aliases":["image_atomic_add_uint"]},{"mnemonic":"image_atomic_add_flt","slug":"image_atomic_add_flt","records":1,"summary":"Add two single-precision float values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_add_flt/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_add_flt.json"},{"mnemonic":"image_atomic_and","slug":"image_atomic_and","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_and/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_and.json"},{"mnemonic":"image_atomic_cmpswap","slug":"image_atomic_cmpswap","records":1,"summary":"Compare two unsigned 32-bit integer values stored in the data comparison register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_cmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_cmpswap.json"},{"mnemonic":"image_atomic_dec","slug":"image_atomic_dec","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in an image surface with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/image_atomic_dec/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_dec.json","aliases":["image_atomic_dec_uint"]},{"mnemonic":"image_atomic_fcmpswap","slug":"image_atomic_fcmpswap","records":1,"summary":"Compare two single-precision float values stored in the data comparison register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_fcmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_fcmpswap.json"},{"mnemonic":"image_atomic_fmax","slug":"image_atomic_fmax","records":1,"summary":"Select the maximum of two single-precision float inputs, given two values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_fmax/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_fmax.json","aliases":["image_atomic_max_flt","image_atomic_max_num_flt"]},{"mnemonic":"image_atomic_fmin","slug":"image_atomic_fmin","records":1,"summary":"Select the minimum of two single-precision float inputs, given two values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_fmin/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_fmin.json","aliases":["image_atomic_min_flt","image_atomic_min_num_flt"]},{"mnemonic":"image_atomic_inc","slug":"image_atomic_inc","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in an image surface with wraparound to 0 if the value exceeds a value in the data register.","page":"https://instructionsets.com/amdgpu/image_atomic_inc/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_inc.json","aliases":["image_atomic_inc_uint"]},{"mnemonic":"image_atomic_max_flt","slug":"image_atomic_max_flt","records":1,"summary":"Select the IEEE maximumNumber() of two single-precision float inputs, given two values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_max_flt/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_max_flt.json","aliases":["image_atomic_fmax","image_atomic_max_num_flt"]},{"mnemonic":"image_atomic_max_num_flt","slug":"image_atomic_max_num_flt","records":1,"summary":"AMDGPU MIMG vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/image_atomic_max_num_flt/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_max_num_flt.json","aliases":["image_atomic_fmax","image_atomic_max_flt"]},{"mnemonic":"image_atomic_min_flt","slug":"image_atomic_min_flt","records":1,"summary":"Select the IEEE minimumNumber() of two single-precision float inputs, given two values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_min_flt/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_min_flt.json","aliases":["image_atomic_fmin","image_atomic_min_num_flt"]},{"mnemonic":"image_atomic_min_num_flt","slug":"image_atomic_min_num_flt","records":1,"summary":"AMDGPU MIMG vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/image_atomic_min_num_flt/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_min_num_flt.json","aliases":["image_atomic_fmin","image_atomic_min_flt"]},{"mnemonic":"image_atomic_or","slug":"image_atomic_or","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_or/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_or.json"},{"mnemonic":"image_atomic_pk_add_bf16","slug":"image_atomic_pk_add_bf16","records":1,"summary":"Add a packed 2-component BF16 float value from the data register to a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_pk_add_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_pk_add_bf16.json"},{"mnemonic":"image_atomic_pk_add_f16","slug":"image_atomic_pk_add_f16","records":1,"summary":"Add a packed 2-component half-precision float value from the data register to a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_pk_add_f16/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_pk_add_f16.json"},{"mnemonic":"image_atomic_rsub","slug":"image_atomic_rsub","records":1,"summary":"AMDGPU MIMG vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/image_atomic_rsub/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_rsub.json"},{"mnemonic":"image_atomic_smax","slug":"image_atomic_smax","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_smax/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_smax.json","aliases":["image_atomic_max_int"]},{"mnemonic":"image_atomic_smin","slug":"image_atomic_smin","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_smin/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_smin.json","aliases":["image_atomic_min_int"]},{"mnemonic":"image_atomic_sub","slug":"image_atomic_sub","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_sub/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_sub.json","aliases":["image_atomic_sub_uint"]},{"mnemonic":"image_atomic_swap","slug":"image_atomic_swap","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_swap/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_swap.json"},{"mnemonic":"image_atomic_umax","slug":"image_atomic_umax","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_umax/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_umax.json","aliases":["image_atomic_max_uint"]},{"mnemonic":"image_atomic_umin","slug":"image_atomic_umin","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_umin/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_umin.json","aliases":["image_atomic_min_uint"]},{"mnemonic":"image_atomic_xor","slug":"image_atomic_xor","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in an image surface.","page":"https://instructionsets.com/amdgpu/image_atomic_xor/","api":"https://instructionsets.com/api/v1/amdgpu/image_atomic_xor.json"},{"mnemonic":"image_bvh64_intersect_ray","slug":"image_bvh64_intersect_ray","records":1,"summary":"Test the intersection of rays with either box nodes or triangle nodes within a bounded volume hierarchy using 64 bit node pointers.","page":"https://instructionsets.com/amdgpu/image_bvh64_intersect_ray/","api":"https://instructionsets.com/api/v1/amdgpu/image_bvh64_intersect_ray.json","aliases":["bvh64_intersect_ray"]},{"mnemonic":"image_bvh8_intersect_ray","slug":"image_bvh8_intersect_ray","records":1,"summary":"This instruction supports testing one BVH8 node against one ray per lane using both intersection engines.","page":"https://instructionsets.com/amdgpu/image_bvh8_intersect_ray/","api":"https://instructionsets.com/api/v1/amdgpu/image_bvh8_intersect_ray.json","aliases":["bvh8_intersect_ray"]},{"mnemonic":"image_bvh_dual_intersect_ray","slug":"image_bvh_dual_intersect_ray","records":1,"summary":"This instruction supports testing two QBVH nodes against the same ray per lane using both intersection engines.","page":"https://instructionsets.com/amdgpu/image_bvh_dual_intersect_ray/","api":"https://instructionsets.com/api/v1/amdgpu/image_bvh_dual_intersect_ray.json","aliases":["bvh_dual_intersect_ray"]},{"mnemonic":"image_bvh_intersect_ray","slug":"image_bvh_intersect_ray","records":1,"summary":"Test the intersection of rays with either box nodes or triangle nodes within a bounded volume hierarchy using 32 bit node pointers.","page":"https://instructionsets.com/amdgpu/image_bvh_intersect_ray/","api":"https://instructionsets.com/api/v1/amdgpu/image_bvh_intersect_ray.json","aliases":["bvh_intersect_ray"]},{"mnemonic":"image_gather4","slug":"image_gather4","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4.json"},{"mnemonic":"image_gather4_b","slug":"image_gather4_b","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_b/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_b.json"},{"mnemonic":"image_gather4_b_cl","slug":"image_gather4_b_cl","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_b_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_b_cl.json"},{"mnemonic":"image_gather4_b_cl_o","slug":"image_gather4_b_cl_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_b_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_b_cl_o.json"},{"mnemonic":"image_gather4_b_o","slug":"image_gather4_b_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_b_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_b_o.json"},{"mnemonic":"image_gather4_c","slug":"image_gather4_c","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c.json"},{"mnemonic":"image_gather4_c_b","slug":"image_gather4_c_b","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_b/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_b.json"},{"mnemonic":"image_gather4_c_b_cl","slug":"image_gather4_c_b_cl","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_b_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_b_cl.json"},{"mnemonic":"image_gather4_c_b_cl_o","slug":"image_gather4_c_b_cl_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_b_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_b_cl_o.json"},{"mnemonic":"image_gather4_c_b_o","slug":"image_gather4_c_b_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_b_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_b_o.json"},{"mnemonic":"image_gather4_c_cl","slug":"image_gather4_c_cl","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_cl.json"},{"mnemonic":"image_gather4_c_cl_o","slug":"image_gather4_c_cl_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_cl_o.json"},{"mnemonic":"image_gather4_c_l","slug":"image_gather4_c_l","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_l/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_l.json"},{"mnemonic":"image_gather4_c_l_o","slug":"image_gather4_c_l_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_l_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_l_o.json"},{"mnemonic":"image_gather4_c_lz","slug":"image_gather4_c_lz","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_lz/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_lz.json"},{"mnemonic":"image_gather4_c_lz_o","slug":"image_gather4_c_lz_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_lz_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_lz_o.json"},{"mnemonic":"image_gather4_c_o","slug":"image_gather4_c_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_c_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_c_o.json"},{"mnemonic":"image_gather4_cl","slug":"image_gather4_cl","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_cl.json"},{"mnemonic":"image_gather4_cl_o","slug":"image_gather4_cl_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_cl_o.json"},{"mnemonic":"image_gather4_l","slug":"image_gather4_l","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_l/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_l.json"},{"mnemonic":"image_gather4_l_o","slug":"image_gather4_l_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_l_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_l_o.json"},{"mnemonic":"image_gather4_lz","slug":"image_gather4_lz","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_lz/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_lz.json"},{"mnemonic":"image_gather4_lz_o","slug":"image_gather4_lz_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_lz_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_lz_o.json"},{"mnemonic":"image_gather4_o","slug":"image_gather4_o","records":1,"summary":"Gather 4 single-component texels from a 2x2 matrix on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4_o.json"},{"mnemonic":"image_gather4h","slug":"image_gather4h","records":1,"summary":"Gather 4 single-component texels from a 4x1 row vector on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4h/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4h.json"},{"mnemonic":"image_gather4h_pck","slug":"image_gather4h_pck","records":1,"summary":"Gather all components of 4 texels from a 4x1 row vector on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather4h_pck/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather4h_pck.json"},{"mnemonic":"image_gather8h_pck","slug":"image_gather8h_pck","records":1,"summary":"Gather all components of 8 texels from a 8x1 row vector on an image surface.","page":"https://instructionsets.com/amdgpu/image_gather8h_pck/","api":"https://instructionsets.com/api/v1/amdgpu/image_gather8h_pck.json"},{"mnemonic":"image_get_lod","slug":"image_get_lod","records":1,"summary":"Return the calculated level of detail (LOD) for the provided input as two single-precision float values. No memory access is performed.","page":"https://instructionsets.com/amdgpu/image_get_lod/","api":"https://instructionsets.com/api/v1/amdgpu/image_get_lod.json"},{"mnemonic":"image_get_resinfo","slug":"image_get_resinfo","records":1,"summary":"Gather resource information for a given miplevel provided in the address register.","page":"https://instructionsets.com/amdgpu/image_get_resinfo/","api":"https://instructionsets.com/api/v1/amdgpu/image_get_resinfo.json"},{"mnemonic":"image_load","slug":"image_load","records":1,"summary":"Load a texel from the largest miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load/","api":"https://instructionsets.com/api/v1/amdgpu/image_load.json"},{"mnemonic":"image_load_by2","slug":"image_load_by2","records":1,"summary":"Load 2 horizontal elements from the largest miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_by2/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_by2.json"},{"mnemonic":"image_load_by4","slug":"image_load_by4","records":1,"summary":"Load 4 horizontal elements from the largest miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_by4/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_by4.json"},{"mnemonic":"image_load_mip","slug":"image_load_mip","records":1,"summary":"Load a texel from a user-specified miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_mip/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_mip.json"},{"mnemonic":"image_load_mip_by2","slug":"image_load_mip_by2","records":1,"summary":"Load 2 horizontal elements from a user-specified miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_mip_by2/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_mip_by2.json"},{"mnemonic":"image_load_mip_by4","slug":"image_load_mip_by4","records":1,"summary":"Load 4 horizontal elements from a user-specified miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_mip_by4/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_mip_by4.json"},{"mnemonic":"image_load_mip_pck","slug":"image_load_mip_pck","records":1,"summary":"Load a texel from a user-specified miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_mip_pck/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_mip_pck.json"},{"mnemonic":"image_load_mip_pck2","slug":"image_load_mip_pck2","records":1,"summary":"Load 2 horizontal elements from a user-specified miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_mip_pck2/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_mip_pck2.json"},{"mnemonic":"image_load_mip_pck4","slug":"image_load_mip_pck4","records":1,"summary":"Load 4 horizontal elements from a user-specified miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_mip_pck4/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_mip_pck4.json"},{"mnemonic":"image_load_mip_pck_sgn","slug":"image_load_mip_pck_sgn","records":1,"summary":"Load a texel from a user-specified miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_mip_pck_sgn/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_mip_pck_sgn.json"},{"mnemonic":"image_load_pck","slug":"image_load_pck","records":1,"summary":"Load a texel from the largest miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_pck/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_pck.json"},{"mnemonic":"image_load_pck2","slug":"image_load_pck2","records":1,"summary":"Load 2 horizontal elements from the largest miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_pck2/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_pck2.json"},{"mnemonic":"image_load_pck4","slug":"image_load_pck4","records":1,"summary":"Load 4 horizontal elements from the largest miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_pck4/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_pck4.json"},{"mnemonic":"image_load_pck_sgn","slug":"image_load_pck_sgn","records":1,"summary":"Load a texel from the largest miplevel in an image surface and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/image_load_pck_sgn/","api":"https://instructionsets.com/api/v1/amdgpu/image_load_pck_sgn.json"},{"mnemonic":"image_msaa_load","slug":"image_msaa_load","records":1,"summary":"Load up to 4 samples of 1 component from an MSAA resource with a user-specified fragment ID. No sampling is performed.","page":"https://instructionsets.com/amdgpu/image_msaa_load/","api":"https://instructionsets.com/api/v1/amdgpu/image_msaa_load.json"},{"mnemonic":"image_sample","slug":"image_sample","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample.json"},{"mnemonic":"image_sample_b","slug":"image_sample_b","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_b/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_b.json"},{"mnemonic":"image_sample_b_cl","slug":"image_sample_b_cl","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_b_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_b_cl.json"},{"mnemonic":"image_sample_b_cl_o","slug":"image_sample_b_cl_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_b_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_b_cl_o.json"},{"mnemonic":"image_sample_b_o","slug":"image_sample_b_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_b_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_b_o.json"},{"mnemonic":"image_sample_c","slug":"image_sample_c","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c.json"},{"mnemonic":"image_sample_c_b","slug":"image_sample_c_b","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_b/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_b.json"},{"mnemonic":"image_sample_c_b_cl","slug":"image_sample_c_b_cl","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_b_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_b_cl.json"},{"mnemonic":"image_sample_c_b_cl_o","slug":"image_sample_c_b_cl_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_b_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_b_cl_o.json"},{"mnemonic":"image_sample_c_b_o","slug":"image_sample_c_b_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_b_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_b_o.json"},{"mnemonic":"image_sample_c_cd","slug":"image_sample_c_cd","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cd/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cd.json"},{"mnemonic":"image_sample_c_cd_cl","slug":"image_sample_c_cd_cl","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cd_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cd_cl.json"},{"mnemonic":"image_sample_c_cd_cl_g16","slug":"image_sample_c_cd_cl_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cd_cl_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cd_cl_g16.json"},{"mnemonic":"image_sample_c_cd_cl_o","slug":"image_sample_c_cd_cl_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cd_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cd_cl_o.json"},{"mnemonic":"image_sample_c_cd_cl_o_g16","slug":"image_sample_c_cd_cl_o_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cd_cl_o_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cd_cl_o_g16.json"},{"mnemonic":"image_sample_c_cd_g16","slug":"image_sample_c_cd_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cd_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cd_g16.json"},{"mnemonic":"image_sample_c_cd_o","slug":"image_sample_c_cd_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cd_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cd_o.json"},{"mnemonic":"image_sample_c_cd_o_g16","slug":"image_sample_c_cd_o_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cd_o_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cd_o_g16.json"},{"mnemonic":"image_sample_c_cl","slug":"image_sample_c_cl","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cl.json"},{"mnemonic":"image_sample_c_cl_o","slug":"image_sample_c_cl_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_cl_o.json"},{"mnemonic":"image_sample_c_d","slug":"image_sample_c_d","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_d/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_d.json"},{"mnemonic":"image_sample_c_d_cl","slug":"image_sample_c_d_cl","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_d_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_d_cl.json"},{"mnemonic":"image_sample_c_d_cl_g16","slug":"image_sample_c_d_cl_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_d_cl_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_d_cl_g16.json"},{"mnemonic":"image_sample_c_d_cl_o","slug":"image_sample_c_d_cl_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_d_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_d_cl_o.json"},{"mnemonic":"image_sample_c_d_cl_o_g16","slug":"image_sample_c_d_cl_o_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_d_cl_o_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_d_cl_o_g16.json"},{"mnemonic":"image_sample_c_d_g16","slug":"image_sample_c_d_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_d_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_d_g16.json"},{"mnemonic":"image_sample_c_d_o","slug":"image_sample_c_d_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_d_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_d_o.json"},{"mnemonic":"image_sample_c_d_o_g16","slug":"image_sample_c_d_o_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_d_o_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_d_o_g16.json"},{"mnemonic":"image_sample_c_l","slug":"image_sample_c_l","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_l/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_l.json"},{"mnemonic":"image_sample_c_l_o","slug":"image_sample_c_l_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_l_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_l_o.json"},{"mnemonic":"image_sample_c_lz","slug":"image_sample_c_lz","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_lz/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_lz.json"},{"mnemonic":"image_sample_c_lz_o","slug":"image_sample_c_lz_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_lz_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_lz_o.json"},{"mnemonic":"image_sample_c_o","slug":"image_sample_c_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_c_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_c_o.json"},{"mnemonic":"image_sample_cd","slug":"image_sample_cd","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cd/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cd.json"},{"mnemonic":"image_sample_cd_cl","slug":"image_sample_cd_cl","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cd_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cd_cl.json"},{"mnemonic":"image_sample_cd_cl_g16","slug":"image_sample_cd_cl_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cd_cl_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cd_cl_g16.json"},{"mnemonic":"image_sample_cd_cl_o","slug":"image_sample_cd_cl_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cd_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cd_cl_o.json"},{"mnemonic":"image_sample_cd_cl_o_g16","slug":"image_sample_cd_cl_o_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cd_cl_o_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cd_cl_o_g16.json"},{"mnemonic":"image_sample_cd_g16","slug":"image_sample_cd_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cd_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cd_g16.json"},{"mnemonic":"image_sample_cd_o","slug":"image_sample_cd_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cd_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cd_o.json"},{"mnemonic":"image_sample_cd_o_g16","slug":"image_sample_cd_o_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cd_o_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cd_o_g16.json"},{"mnemonic":"image_sample_cl","slug":"image_sample_cl","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cl.json"},{"mnemonic":"image_sample_cl_o","slug":"image_sample_cl_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_cl_o.json"},{"mnemonic":"image_sample_d","slug":"image_sample_d","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_d/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_d.json"},{"mnemonic":"image_sample_d_cl","slug":"image_sample_d_cl","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_d_cl/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_d_cl.json"},{"mnemonic":"image_sample_d_cl_g16","slug":"image_sample_d_cl_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_d_cl_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_d_cl_g16.json"},{"mnemonic":"image_sample_d_cl_o","slug":"image_sample_d_cl_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_d_cl_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_d_cl_o.json"},{"mnemonic":"image_sample_d_cl_o_g16","slug":"image_sample_d_cl_o_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_d_cl_o_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_d_cl_o_g16.json"},{"mnemonic":"image_sample_d_g16","slug":"image_sample_d_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_d_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_d_g16.json"},{"mnemonic":"image_sample_d_o","slug":"image_sample_d_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_d_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_d_o.json"},{"mnemonic":"image_sample_d_o_g16","slug":"image_sample_d_o_g16","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_d_o_g16/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_d_o_g16.json"},{"mnemonic":"image_sample_l","slug":"image_sample_l","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_l/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_l.json"},{"mnemonic":"image_sample_l_o","slug":"image_sample_l_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_l_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_l_o.json"},{"mnemonic":"image_sample_lz","slug":"image_sample_lz","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_lz/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_lz.json"},{"mnemonic":"image_sample_lz_o","slug":"image_sample_lz_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_lz_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_lz_o.json"},{"mnemonic":"image_sample_o","slug":"image_sample_o","records":1,"summary":"Sample texels from an image surface using texel coordinates provided by the address input registers and store the result into vector registers.","page":"https://instructionsets.com/amdgpu/image_sample_o/","api":"https://instructionsets.com/api/v1/amdgpu/image_sample_o.json"},{"mnemonic":"image_store","slug":"image_store","records":1,"summary":"Store a texel from a vector register to the largest miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store/","api":"https://instructionsets.com/api/v1/amdgpu/image_store.json"},{"mnemonic":"image_store_by2","slug":"image_store_by2","records":1,"summary":"Store 2 horizontal elements from a vector register to the largest miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_by2/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_by2.json"},{"mnemonic":"image_store_by4","slug":"image_store_by4","records":1,"summary":"Store 4 horizontal elements from a vector register to the largest miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_by4/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_by4.json"},{"mnemonic":"image_store_mip","slug":"image_store_mip","records":1,"summary":"Store a texel from a vector register to a user-specified miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_mip/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_mip.json"},{"mnemonic":"image_store_mip_by2","slug":"image_store_mip_by2","records":1,"summary":"Store 2 horizontal elements from a vector register to a user-specified miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_mip_by2/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_mip_by2.json"},{"mnemonic":"image_store_mip_by4","slug":"image_store_mip_by4","records":1,"summary":"Store 4 horizontal elements from a vector register to a user-specified miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_mip_by4/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_mip_by4.json"},{"mnemonic":"image_store_mip_pck","slug":"image_store_mip_pck","records":1,"summary":"Store a texel from a vector register to a user-specified miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_mip_pck/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_mip_pck.json"},{"mnemonic":"image_store_mip_pck2","slug":"image_store_mip_pck2","records":1,"summary":"Store 2 horizontal elements from a vector register to a user-specified miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_mip_pck2/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_mip_pck2.json"},{"mnemonic":"image_store_mip_pck4","slug":"image_store_mip_pck4","records":1,"summary":"Store 4 horizontal elements from a vector register to a user-specified miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_mip_pck4/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_mip_pck4.json"},{"mnemonic":"image_store_pck","slug":"image_store_pck","records":1,"summary":"Store a texel from a vector register to the largest miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_pck/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_pck.json"},{"mnemonic":"image_store_pck2","slug":"image_store_pck2","records":1,"summary":"Store 2 horizontal elements from a vector register to the largest miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_pck2/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_pck2.json"},{"mnemonic":"image_store_pck4","slug":"image_store_pck4","records":1,"summary":"Store 4 horizontal elements from a vector register to the largest miplevel in an image surface.","page":"https://instructionsets.com/amdgpu/image_store_pck4/","api":"https://instructionsets.com/api/v1/amdgpu/image_store_pck4.json"},{"mnemonic":"lds_direct_load","slug":"lds_direct_load","records":1,"summary":"Read a single 32-bit value from LDS to all lanes.","page":"https://instructionsets.com/amdgpu/lds_direct_load/","api":"https://instructionsets.com/api/v1/amdgpu/lds_direct_load.json","aliases":["ds_direct_load"]},{"mnemonic":"lds_param_load","slug":"lds_param_load","records":1,"summary":"Transfer parameter data from LDS to VGPRs and expand data in LDS using the NewPrimMask (provided in M0) to place per-quad data into lanes 0-3 of each…","page":"https://instructionsets.com/amdgpu/lds_param_load/","api":"https://instructionsets.com/api/v1/amdgpu/lds_param_load.json","aliases":["ds_param_load"]},{"mnemonic":"s_abs_i32","slug":"s_abs_i32","records":1,"summary":"Compute the absolute value of a scalar input, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_abs_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_abs_i32.json"},{"mnemonic":"s_absdiff_i32","slug":"s_absdiff_i32","records":1,"summary":"Calculate the absolute value of difference between two scalar inputs, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_absdiff_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_absdiff_i32.json"},{"mnemonic":"s_add_co_ci_u32","slug":"s_add_co_ci_u32","records":1,"summary":"Add two unsigned 32-bit integer inputs and a carry-in bit from SCC, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_add_co_ci_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_co_ci_u32.json","aliases":["s_addc_u32"]},{"mnemonic":"s_add_co_i32","slug":"s_add_co_i32","records":1,"summary":"Add two signed 32-bit integer inputs, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_add_co_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_co_i32.json","aliases":["s_add_i32"]},{"mnemonic":"s_add_co_u32","slug":"s_add_co_u32","records":1,"summary":"Add two unsigned 32-bit integer inputs, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_add_co_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_co_u32.json","aliases":["s_add_u32"]},{"mnemonic":"s_add_f16","slug":"s_add_f16","records":1,"summary":"Add two floating point inputs and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_add_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_f16.json"},{"mnemonic":"s_add_f32","slug":"s_add_f32","records":1,"summary":"Add two floating point inputs and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_add_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_f32.json"},{"mnemonic":"s_add_i32","slug":"s_add_i32","records":1,"summary":"Add two signed 32-bit integer inputs, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_add_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_i32.json","aliases":["s_add_co_i32"]},{"mnemonic":"s_add_nc_u64","slug":"s_add_nc_u64","records":1,"summary":"Add two unsigned 64-bit integer inputs and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_add_nc_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_nc_u64.json","aliases":["s_add_u64"]},{"mnemonic":"s_add_pc_i64","slug":"s_add_pc_i64","records":1,"summary":"AMDGPU SOP1 scalar instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_add_pc_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_pc_i64.json"},{"mnemonic":"s_add_u32","slug":"s_add_u32","records":1,"summary":"Add two 32-bit unsigned scalar operands, wavefront-uniform.","page":"https://instructionsets.com/amdgpu/s_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_u32.json","aliases":["s_add_co_u32"]},{"mnemonic":"s_add_u64","slug":"s_add_u64","records":1,"summary":"AMDGPU SOP2 scalar instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_add_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_add_u64.json","aliases":["s_add_nc_u64"]},{"mnemonic":"s_addc_u32","slug":"s_addc_u32","records":1,"summary":"Add two unsigned 32-bit integer inputs and a carry-in bit from SCC, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_addc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_addc_u32.json","aliases":["s_add_co_ci_u32"]},{"mnemonic":"s_addk_co_i32","slug":"s_addk_co_i32","records":1,"summary":"Add a scalar input and the sign extension of a literal 16-bit constant, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_addk_co_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_addk_co_i32.json","aliases":["s_addk_i32"]},{"mnemonic":"s_addk_i32","slug":"s_addk_i32","records":1,"summary":"Add a scalar input and the sign extension of a literal 16-bit constant, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_addk_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_addk_i32.json","aliases":["s_addk_co_i32"]},{"mnemonic":"s_alloc_vgpr","slug":"s_alloc_vgpr","records":1,"summary":"Attempt to set the wave's VGPR allocation to the specified number of VGPRs (or greater).","page":"https://instructionsets.com/amdgpu/s_alloc_vgpr/","api":"https://instructionsets.com/api/v1/amdgpu/s_alloc_vgpr.json"},{"mnemonic":"s_and_b32","slug":"s_and_b32","records":1,"summary":"Calculate bitwise AND on two scalar inputs, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_and_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_b32.json"},{"mnemonic":"s_and_b64","slug":"s_and_b64","records":1,"summary":"Calculate bitwise AND on two scalar inputs, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_and_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_b64.json"},{"mnemonic":"s_and_not0_saveexec_b32","slug":"s_and_not0_saveexec_b32","records":1,"summary":"Calculate bitwise AND on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_and_not0_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not0_saveexec_b32.json","aliases":["s_andn1_saveexec_b32"]},{"mnemonic":"s_and_not0_saveexec_b64","slug":"s_and_not0_saveexec_b64","records":1,"summary":"Calculate bitwise AND on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_and_not0_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not0_saveexec_b64.json","aliases":["s_andn1_saveexec_b64"]},{"mnemonic":"s_and_not0_wrexec_b32","slug":"s_and_not0_wrexec_b32","records":1,"summary":"Calculate bitwise AND on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask and also into the scalar…","page":"https://instructionsets.com/amdgpu/s_and_not0_wrexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not0_wrexec_b32.json","aliases":["s_andn1_wrexec_b32"]},{"mnemonic":"s_and_not0_wrexec_b64","slug":"s_and_not0_wrexec_b64","records":1,"summary":"Calculate bitwise AND on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask and also into the scalar…","page":"https://instructionsets.com/amdgpu/s_and_not0_wrexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not0_wrexec_b64.json","aliases":["s_andn1_wrexec_b64"]},{"mnemonic":"s_and_not1_b32","slug":"s_and_not1_b32","records":1,"summary":"Calculate bitwise AND with the first input and the negation of the second input, store the result into a scalar register and set SCC if the result is…","page":"https://instructionsets.com/amdgpu/s_and_not1_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not1_b32.json","aliases":["s_andn2_b32"]},{"mnemonic":"s_and_not1_b64","slug":"s_and_not1_b64","records":1,"summary":"Calculate bitwise AND with the first input and the negation of the second input, store the result into a scalar register and set SCC if the result is…","page":"https://instructionsets.com/amdgpu/s_and_not1_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not1_b64.json","aliases":["s_andn2_b64"]},{"mnemonic":"s_and_not1_saveexec_b32","slug":"s_and_not1_saveexec_b32","records":1,"summary":"Calculate bitwise AND on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_and_not1_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not1_saveexec_b32.json","aliases":["s_andn2_saveexec_b32"]},{"mnemonic":"s_and_not1_saveexec_b64","slug":"s_and_not1_saveexec_b64","records":1,"summary":"Calculate bitwise AND on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_and_not1_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not1_saveexec_b64.json","aliases":["s_andn2_saveexec_b64"]},{"mnemonic":"s_and_not1_wrexec_b32","slug":"s_and_not1_wrexec_b32","records":1,"summary":"Calculate bitwise AND on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask and also into the scalar…","page":"https://instructionsets.com/amdgpu/s_and_not1_wrexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not1_wrexec_b32.json","aliases":["s_andn2_wrexec_b32"]},{"mnemonic":"s_and_not1_wrexec_b64","slug":"s_and_not1_wrexec_b64","records":1,"summary":"Calculate bitwise AND on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask and also into the scalar…","page":"https://instructionsets.com/amdgpu/s_and_not1_wrexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_not1_wrexec_b64.json","aliases":["s_andn2_wrexec_b64"]},{"mnemonic":"s_and_saveexec_b32","slug":"s_and_saveexec_b32","records":1,"summary":"Calculate bitwise AND on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_and_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_saveexec_b32.json"},{"mnemonic":"s_and_saveexec_b64","slug":"s_and_saveexec_b64","records":1,"summary":"Calculate bitwise AND on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_and_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_and_saveexec_b64.json"},{"mnemonic":"s_andn1_saveexec_b32","slug":"s_andn1_saveexec_b32","records":1,"summary":"Calculate bitwise AND on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_andn1_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn1_saveexec_b32.json","aliases":["s_and_not0_saveexec_b32"]},{"mnemonic":"s_andn1_saveexec_b64","slug":"s_andn1_saveexec_b64","records":1,"summary":"Calculate bitwise AND on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_andn1_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn1_saveexec_b64.json","aliases":["s_and_not0_saveexec_b64"]},{"mnemonic":"s_andn1_wrexec_b32","slug":"s_andn1_wrexec_b32","records":1,"summary":"Calculate bitwise AND on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask and also into the scalar…","page":"https://instructionsets.com/amdgpu/s_andn1_wrexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn1_wrexec_b32.json","aliases":["s_and_not0_wrexec_b32"]},{"mnemonic":"s_andn1_wrexec_b64","slug":"s_andn1_wrexec_b64","records":1,"summary":"Calculate bitwise AND on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask and also into the scalar…","page":"https://instructionsets.com/amdgpu/s_andn1_wrexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn1_wrexec_b64.json","aliases":["s_and_not0_wrexec_b64"]},{"mnemonic":"s_andn2_b32","slug":"s_andn2_b32","records":1,"summary":"Calculate bitwise AND with the first input and the negation of the second input, store the result into a scalar register and set SCC if the result is…","page":"https://instructionsets.com/amdgpu/s_andn2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn2_b32.json","aliases":["s_and_not1_b32"]},{"mnemonic":"s_andn2_b64","slug":"s_andn2_b64","records":1,"summary":"Calculate bitwise AND with the first input and the negation of the second input, store the result into a scalar register and set SCC if the result is…","page":"https://instructionsets.com/amdgpu/s_andn2_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn2_b64.json","aliases":["s_and_not1_b64"]},{"mnemonic":"s_andn2_saveexec_b32","slug":"s_andn2_saveexec_b32","records":1,"summary":"Calculate bitwise AND on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_andn2_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn2_saveexec_b32.json","aliases":["s_and_not1_saveexec_b32"]},{"mnemonic":"s_andn2_saveexec_b64","slug":"s_andn2_saveexec_b64","records":1,"summary":"Calculate bitwise AND on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_andn2_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn2_saveexec_b64.json","aliases":["s_and_not1_saveexec_b64"]},{"mnemonic":"s_andn2_wrexec_b32","slug":"s_andn2_wrexec_b32","records":1,"summary":"Calculate bitwise AND on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask and also into the scalar…","page":"https://instructionsets.com/amdgpu/s_andn2_wrexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn2_wrexec_b32.json","aliases":["s_and_not1_wrexec_b32"]},{"mnemonic":"s_andn2_wrexec_b64","slug":"s_andn2_wrexec_b64","records":1,"summary":"Calculate bitwise AND on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask and also into the scalar…","page":"https://instructionsets.com/amdgpu/s_andn2_wrexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_andn2_wrexec_b64.json","aliases":["s_and_not1_wrexec_b64"]},{"mnemonic":"s_ashr_i32","slug":"s_ashr_i32","records":1,"summary":"Given a shift count in the second scalar input, calculate the arithmetic shift right (preserving sign bit) of the first scalar input, store the…","page":"https://instructionsets.com/amdgpu/s_ashr_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_ashr_i32.json"},{"mnemonic":"s_ashr_i64","slug":"s_ashr_i64","records":1,"summary":"Given a shift count in the second scalar input, calculate the arithmetic shift right (preserving sign bit) of the first scalar input, store the…","page":"https://instructionsets.com/amdgpu/s_ashr_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_ashr_i64.json"},{"mnemonic":"s_atc_probe","slug":"s_atc_probe","records":1,"summary":"Probe or prefetch an address into the scalar data cache.","page":"https://instructionsets.com/amdgpu/s_atc_probe/","api":"https://instructionsets.com/api/v1/amdgpu/s_atc_probe.json"},{"mnemonic":"s_atc_probe_buffer","slug":"s_atc_probe_buffer","records":1,"summary":"Probe or prefetch an address into the scalar data cache.","page":"https://instructionsets.com/amdgpu/s_atc_probe_buffer/","api":"https://instructionsets.com/api/v1/amdgpu/s_atc_probe_buffer.json"},{"mnemonic":"s_atomic_add","slug":"s_atomic_add","records":1,"summary":"Add two unsigned 32-bit integer values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_add/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_add.json"},{"mnemonic":"s_atomic_add_x2","slug":"s_atomic_add_x2","records":1,"summary":"Add two unsigned 64-bit integer values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_add_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_add_x2.json"},{"mnemonic":"s_atomic_and","slug":"s_atomic_and","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_and/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_and.json"},{"mnemonic":"s_atomic_and_x2","slug":"s_atomic_and_x2","records":1,"summary":"Calculate bitwise AND given two unsigned 64-bit integer values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_and_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_and_x2.json"},{"mnemonic":"s_atomic_cmpswap","slug":"s_atomic_cmpswap","records":1,"summary":"Compare two unsigned 32-bit integer values stored in the data comparison register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_cmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_cmpswap.json"},{"mnemonic":"s_atomic_cmpswap_x2","slug":"s_atomic_cmpswap_x2","records":1,"summary":"Compare two unsigned 64-bit integer values stored in the data comparison register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_cmpswap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_cmpswap_x2.json"},{"mnemonic":"s_atomic_dec","slug":"s_atomic_dec","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in the scalar memory with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/s_atomic_dec/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_dec.json"},{"mnemonic":"s_atomic_dec_x2","slug":"s_atomic_dec_x2","records":1,"summary":"Decrement an unsigned 64-bit integer value from a location in the scalar memory with wraparound to a value in the data register if the decrement…","page":"https://instructionsets.com/amdgpu/s_atomic_dec_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_dec_x2.json"},{"mnemonic":"s_atomic_inc","slug":"s_atomic_inc","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in the scalar memory with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/s_atomic_inc/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_inc.json"},{"mnemonic":"s_atomic_inc_x2","slug":"s_atomic_inc_x2","records":1,"summary":"Increment an unsigned 64-bit integer value from a location in the scalar memory with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/s_atomic_inc_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_inc_x2.json"},{"mnemonic":"s_atomic_or","slug":"s_atomic_or","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_or/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_or.json"},{"mnemonic":"s_atomic_or_x2","slug":"s_atomic_or_x2","records":1,"summary":"Calculate bitwise OR given two unsigned 64-bit integer values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_or_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_or_x2.json"},{"mnemonic":"s_atomic_smax","slug":"s_atomic_smax","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_smax/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_smax.json"},{"mnemonic":"s_atomic_smax_x2","slug":"s_atomic_smax_x2","records":1,"summary":"Select the maximum of two signed 64-bit integer inputs, given two values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_smax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_smax_x2.json"},{"mnemonic":"s_atomic_smin","slug":"s_atomic_smin","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_smin/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_smin.json"},{"mnemonic":"s_atomic_smin_x2","slug":"s_atomic_smin_x2","records":1,"summary":"Select the minimum of two signed 64-bit integer inputs, given two values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_smin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_smin_x2.json"},{"mnemonic":"s_atomic_sub","slug":"s_atomic_sub","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_sub/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_sub.json"},{"mnemonic":"s_atomic_sub_x2","slug":"s_atomic_sub_x2","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in the data register from a value stored in a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_sub_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_sub_x2.json"},{"mnemonic":"s_atomic_swap","slug":"s_atomic_swap","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_swap/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_swap.json"},{"mnemonic":"s_atomic_swap_x2","slug":"s_atomic_swap_x2","records":1,"summary":"Swap an unsigned 64-bit integer value in the data register with a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_swap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_swap_x2.json"},{"mnemonic":"s_atomic_umax","slug":"s_atomic_umax","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_umax/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_umax.json"},{"mnemonic":"s_atomic_umax_x2","slug":"s_atomic_umax_x2","records":1,"summary":"Select the maximum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_umax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_umax_x2.json"},{"mnemonic":"s_atomic_umin","slug":"s_atomic_umin","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_umin/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_umin.json"},{"mnemonic":"s_atomic_umin_x2","slug":"s_atomic_umin_x2","records":1,"summary":"Select the minimum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_umin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_umin_x2.json"},{"mnemonic":"s_atomic_xor","slug":"s_atomic_xor","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_xor/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_xor.json"},{"mnemonic":"s_atomic_xor_x2","slug":"s_atomic_xor_x2","records":1,"summary":"Calculate bitwise XOR given two unsigned 64-bit integer values stored in the data register and a location in the scalar memory.","page":"https://instructionsets.com/amdgpu/s_atomic_xor_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_atomic_xor_x2.json"},{"mnemonic":"s_barrier","slug":"s_barrier","records":1,"summary":"Synchronize all waves of the executing workgroup at this point.","page":"https://instructionsets.com/amdgpu/s_barrier/","api":"https://instructionsets.com/api/v1/amdgpu/s_barrier.json"},{"mnemonic":"s_barrier_init","slug":"s_barrier_init","records":1,"summary":"AMDGPU SOP1 scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_barrier_init/","api":"https://instructionsets.com/api/v1/amdgpu/s_barrier_init.json"},{"mnemonic":"s_barrier_join","slug":"s_barrier_join","records":1,"summary":"AMDGPU SOP1 scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_barrier_join/","api":"https://instructionsets.com/api/v1/amdgpu/s_barrier_join.json"},{"mnemonic":"s_barrier_leave","slug":"s_barrier_leave","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_barrier_leave/","api":"https://instructionsets.com/api/v1/amdgpu/s_barrier_leave.json"},{"mnemonic":"s_barrier_signal","slug":"s_barrier_signal","records":1,"summary":"Signal that a wave has arrived at a barrier . The argument specifies which barrier to signal.","page":"https://instructionsets.com/amdgpu/s_barrier_signal/","api":"https://instructionsets.com/api/v1/amdgpu/s_barrier_signal.json"},{"mnemonic":"s_barrier_signal_isfirst","slug":"s_barrier_signal_isfirst","records":1,"summary":"Signal that a wave has arrived at a barrier and set SCC to indicate if this is the first wave to signal the barrier.","page":"https://instructionsets.com/amdgpu/s_barrier_signal_isfirst/","api":"https://instructionsets.com/api/v1/amdgpu/s_barrier_signal_isfirst.json"},{"mnemonic":"s_barrier_wait","slug":"s_barrier_wait","records":1,"summary":"Wait for a barrier to complete. The SIMM16 argument specifies which barrier to wait on.","page":"https://instructionsets.com/amdgpu/s_barrier_wait/","api":"https://instructionsets.com/api/v1/amdgpu/s_barrier_wait.json"},{"mnemonic":"s_bcnt0_i32_b32","slug":"s_bcnt0_i32_b32","records":1,"summary":"Count the number of \"0\" bits in a scalar input, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_bcnt0_i32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bcnt0_i32_b32.json"},{"mnemonic":"s_bcnt0_i32_b64","slug":"s_bcnt0_i32_b64","records":1,"summary":"Count the number of \"0\" bits in a scalar input, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_bcnt0_i32_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_bcnt0_i32_b64.json"},{"mnemonic":"s_bcnt1_i32_b32","slug":"s_bcnt1_i32_b32","records":1,"summary":"Count the number of \"1\" bits in a scalar input, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_bcnt1_i32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bcnt1_i32_b32.json"},{"mnemonic":"s_bcnt1_i32_b64","slug":"s_bcnt1_i32_b64","records":1,"summary":"Count the number of \"1\" bits in a scalar input, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_bcnt1_i32_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_bcnt1_i32_b64.json"},{"mnemonic":"s_bfe_i32","slug":"s_bfe_i32","records":1,"summary":"Extract a signed bitfield from the first input using field offset and size encoded in the second input, store the result into a scalar register and…","page":"https://instructionsets.com/amdgpu/s_bfe_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bfe_i32.json"},{"mnemonic":"s_bfe_i64","slug":"s_bfe_i64","records":1,"summary":"Extract a signed bitfield from the first input using field offset and size encoded in the second input, store the result into a scalar register and…","page":"https://instructionsets.com/amdgpu/s_bfe_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_bfe_i64.json"},{"mnemonic":"s_bfe_u32","slug":"s_bfe_u32","records":1,"summary":"Extract an unsigned bitfield from the first input using field offset and size encoded in the second input, store the result into a scalar register…","page":"https://instructionsets.com/amdgpu/s_bfe_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bfe_u32.json"},{"mnemonic":"s_bfe_u64","slug":"s_bfe_u64","records":1,"summary":"Extract an unsigned bitfield from the first input using field offset and size encoded in the second input, store the result into a scalar register…","page":"https://instructionsets.com/amdgpu/s_bfe_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_bfe_u64.json"},{"mnemonic":"s_bfm_b32","slug":"s_bfm_b32","records":1,"summary":"Calculate a bitfield mask given a field offset and size and store the result in a scalar register.","page":"https://instructionsets.com/amdgpu/s_bfm_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bfm_b32.json"},{"mnemonic":"s_bfm_b64","slug":"s_bfm_b64","records":1,"summary":"Calculate a bitfield mask given a field offset and size and store the result in a scalar register.","page":"https://instructionsets.com/amdgpu/s_bfm_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_bfm_b64.json"},{"mnemonic":"s_bitcmp0_b32","slug":"s_bitcmp0_b32","records":1,"summary":"Extract a bit from the first scalar input based on an index in the second scalar input, and set SCC to 1 iff the extracted bit is equal to 0.","page":"https://instructionsets.com/amdgpu/s_bitcmp0_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bitcmp0_b32.json"},{"mnemonic":"s_bitcmp0_b64","slug":"s_bitcmp0_b64","records":1,"summary":"Extract a bit from the first scalar input based on an index in the second scalar input, and set SCC to 1 iff the extracted bit is equal to 0.","page":"https://instructionsets.com/amdgpu/s_bitcmp0_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_bitcmp0_b64.json"},{"mnemonic":"s_bitcmp1_b32","slug":"s_bitcmp1_b32","records":1,"summary":"Extract a bit from the first scalar input based on an index in the second scalar input, and set SCC to 1 iff the extracted bit is equal to 1.","page":"https://instructionsets.com/amdgpu/s_bitcmp1_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bitcmp1_b32.json"},{"mnemonic":"s_bitcmp1_b64","slug":"s_bitcmp1_b64","records":1,"summary":"Extract a bit from the first scalar input based on an index in the second scalar input, and set SCC to 1 iff the extracted bit is equal to 1.","page":"https://instructionsets.com/amdgpu/s_bitcmp1_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_bitcmp1_b64.json"},{"mnemonic":"s_bitreplicate_b64_b32","slug":"s_bitreplicate_b64_b32","records":1,"summary":"Substitute each bit of a 32 bit scalar input with two instances of itself and store the result into a 64 bit scalar register.","page":"https://instructionsets.com/amdgpu/s_bitreplicate_b64_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bitreplicate_b64_b32.json"},{"mnemonic":"s_bitset0_b32","slug":"s_bitset0_b32","records":1,"summary":"Given a bit offset in a scalar input, set the indicated bit in the destination scalar register to 0.","page":"https://instructionsets.com/amdgpu/s_bitset0_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bitset0_b32.json"},{"mnemonic":"s_bitset0_b64","slug":"s_bitset0_b64","records":1,"summary":"Given a bit offset in a scalar input, set the indicated bit in the destination scalar register to 0.","page":"https://instructionsets.com/amdgpu/s_bitset0_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_bitset0_b64.json"},{"mnemonic":"s_bitset1_b32","slug":"s_bitset1_b32","records":1,"summary":"Given a bit offset in a scalar input, set the indicated bit in the destination scalar register to 1.","page":"https://instructionsets.com/amdgpu/s_bitset1_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_bitset1_b32.json"},{"mnemonic":"s_bitset1_b64","slug":"s_bitset1_b64","records":1,"summary":"Given a bit offset in a scalar input, set the indicated bit in the destination scalar register to 1.","page":"https://instructionsets.com/amdgpu/s_bitset1_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_bitset1_b64.json"},{"mnemonic":"s_branch","slug":"s_branch","records":1,"summary":"Unconditional relative branch.","page":"https://instructionsets.com/amdgpu/s_branch/","api":"https://instructionsets.com/api/v1/amdgpu/s_branch.json"},{"mnemonic":"s_brev_b32","slug":"s_brev_b32","records":1,"summary":"Reverse the order of bits in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_brev_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_brev_b32.json"},{"mnemonic":"s_brev_b64","slug":"s_brev_b64","records":1,"summary":"Reverse the order of bits in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_brev_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_brev_b64.json"},{"mnemonic":"s_buffer_atomic_add","slug":"s_buffer_atomic_add","records":1,"summary":"Add two unsigned 32-bit integer values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_add/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_add.json"},{"mnemonic":"s_buffer_atomic_add_x2","slug":"s_buffer_atomic_add_x2","records":1,"summary":"Add two unsigned 64-bit integer values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_add_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_add_x2.json"},{"mnemonic":"s_buffer_atomic_and","slug":"s_buffer_atomic_and","records":1,"summary":"Calculate bitwise AND given two unsigned 32-bit integer values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_and/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_and.json"},{"mnemonic":"s_buffer_atomic_and_x2","slug":"s_buffer_atomic_and_x2","records":1,"summary":"Calculate bitwise AND given two unsigned 64-bit integer values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_and_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_and_x2.json"},{"mnemonic":"s_buffer_atomic_cmpswap","slug":"s_buffer_atomic_cmpswap","records":1,"summary":"Compare two unsigned 32-bit integer values stored in the data comparison register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_cmpswap/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_cmpswap.json"},{"mnemonic":"s_buffer_atomic_cmpswap_x2","slug":"s_buffer_atomic_cmpswap_x2","records":1,"summary":"Compare two unsigned 64-bit integer values stored in the data comparison register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_cmpswap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_cmpswap_x2.json"},{"mnemonic":"s_buffer_atomic_dec","slug":"s_buffer_atomic_dec","records":1,"summary":"Decrement an unsigned 32-bit integer value from a location in a scalar buffer surface with wraparound to a value in the data register if the…","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_dec/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_dec.json"},{"mnemonic":"s_buffer_atomic_dec_x2","slug":"s_buffer_atomic_dec_x2","records":1,"summary":"Decrement an unsigned 64-bit integer value from a location in a scalar buffer surface with wraparound to a value in the data register if the…","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_dec_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_dec_x2.json"},{"mnemonic":"s_buffer_atomic_inc","slug":"s_buffer_atomic_inc","records":1,"summary":"Increment an unsigned 32-bit integer value from a location in a scalar buffer surface with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_inc/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_inc.json"},{"mnemonic":"s_buffer_atomic_inc_x2","slug":"s_buffer_atomic_inc_x2","records":1,"summary":"Increment an unsigned 64-bit integer value from a location in a scalar buffer surface with wraparound to 0 if the value exceeds a value in the data…","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_inc_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_inc_x2.json"},{"mnemonic":"s_buffer_atomic_or","slug":"s_buffer_atomic_or","records":1,"summary":"Calculate bitwise OR given two unsigned 32-bit integer values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_or/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_or.json"},{"mnemonic":"s_buffer_atomic_or_x2","slug":"s_buffer_atomic_or_x2","records":1,"summary":"Calculate bitwise OR given two unsigned 64-bit integer values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_or_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_or_x2.json"},{"mnemonic":"s_buffer_atomic_smax","slug":"s_buffer_atomic_smax","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_smax/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_smax.json"},{"mnemonic":"s_buffer_atomic_smax_x2","slug":"s_buffer_atomic_smax_x2","records":1,"summary":"Select the maximum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_smax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_smax_x2.json"},{"mnemonic":"s_buffer_atomic_smin","slug":"s_buffer_atomic_smin","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, given two values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_smin/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_smin.json"},{"mnemonic":"s_buffer_atomic_smin_x2","slug":"s_buffer_atomic_smin_x2","records":1,"summary":"Select the minimum of two signed 64-bit integer inputs, given two values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_smin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_smin_x2.json"},{"mnemonic":"s_buffer_atomic_sub","slug":"s_buffer_atomic_sub","records":1,"summary":"Subtract an unsigned 32-bit integer value stored in the data register from a value stored in a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_sub/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_sub.json"},{"mnemonic":"s_buffer_atomic_sub_x2","slug":"s_buffer_atomic_sub_x2","records":1,"summary":"Subtract an unsigned 64-bit integer value stored in the data register from a value stored in a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_sub_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_sub_x2.json"},{"mnemonic":"s_buffer_atomic_swap","slug":"s_buffer_atomic_swap","records":1,"summary":"Swap an unsigned 32-bit integer value in the data register with a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_swap/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_swap.json"},{"mnemonic":"s_buffer_atomic_swap_x2","slug":"s_buffer_atomic_swap_x2","records":1,"summary":"Swap an unsigned 64-bit integer value in the data register with a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_swap_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_swap_x2.json"},{"mnemonic":"s_buffer_atomic_umax","slug":"s_buffer_atomic_umax","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_umax/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_umax.json"},{"mnemonic":"s_buffer_atomic_umax_x2","slug":"s_buffer_atomic_umax_x2","records":1,"summary":"Select the maximum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_umax_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_umax_x2.json"},{"mnemonic":"s_buffer_atomic_umin","slug":"s_buffer_atomic_umin","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, given two values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_umin/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_umin.json"},{"mnemonic":"s_buffer_atomic_umin_x2","slug":"s_buffer_atomic_umin_x2","records":1,"summary":"Select the minimum of two unsigned 64-bit integer inputs, given two values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_umin_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_umin_x2.json"},{"mnemonic":"s_buffer_atomic_xor","slug":"s_buffer_atomic_xor","records":1,"summary":"Calculate bitwise XOR given two unsigned 32-bit integer values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_xor/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_xor.json"},{"mnemonic":"s_buffer_atomic_xor_x2","slug":"s_buffer_atomic_xor_x2","records":1,"summary":"Calculate bitwise XOR given two unsigned 64-bit integer values stored in the data register and a location in a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_atomic_xor_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_atomic_xor_x2.json"},{"mnemonic":"s_buffer_load_b128","slug":"s_buffer_load_b128","records":1,"summary":"Load 128 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_b128/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_b128.json","aliases":["s_buffer_load_dwordx4"]},{"mnemonic":"s_buffer_load_b256","slug":"s_buffer_load_b256","records":1,"summary":"Load 256 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_b256/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_b256.json","aliases":["s_buffer_load_dwordx8"]},{"mnemonic":"s_buffer_load_b32","slug":"s_buffer_load_b32","records":1,"summary":"Load 32 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_b32.json","aliases":["s_buffer_load_dword"]},{"mnemonic":"s_buffer_load_b512","slug":"s_buffer_load_b512","records":1,"summary":"Load 512 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_b512/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_b512.json","aliases":["s_buffer_load_dwordx16"]},{"mnemonic":"s_buffer_load_b64","slug":"s_buffer_load_b64","records":1,"summary":"Load 64 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_b64.json","aliases":["s_buffer_load_dwordx2"]},{"mnemonic":"s_buffer_load_b96","slug":"s_buffer_load_b96","records":1,"summary":"Load 96 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_b96/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_b96.json"},{"mnemonic":"s_buffer_load_dword","slug":"s_buffer_load_dword","records":1,"summary":"Load 32 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_dword/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_dword.json","aliases":["s_buffer_load_b32"]},{"mnemonic":"s_buffer_load_dwordx16","slug":"s_buffer_load_dwordx16","records":1,"summary":"Load 512 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_dwordx16/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_dwordx16.json","aliases":["s_buffer_load_b512"]},{"mnemonic":"s_buffer_load_dwordx2","slug":"s_buffer_load_dwordx2","records":1,"summary":"Load 64 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_dwordx2.json","aliases":["s_buffer_load_b64"]},{"mnemonic":"s_buffer_load_dwordx4","slug":"s_buffer_load_dwordx4","records":1,"summary":"Load 128 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_dwordx4.json","aliases":["s_buffer_load_b128"]},{"mnemonic":"s_buffer_load_dwordx8","slug":"s_buffer_load_dwordx8","records":1,"summary":"Load 256 bits of data from a scalar buffer surface into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_dwordx8/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_dwordx8.json","aliases":["s_buffer_load_b256"]},{"mnemonic":"s_buffer_load_i16","slug":"s_buffer_load_i16","records":1,"summary":"Load 16 bits of signed data from a scalar buffer surface, sign extend to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_i16/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_i16.json"},{"mnemonic":"s_buffer_load_i8","slug":"s_buffer_load_i8","records":1,"summary":"Load 8 bits of signed data from a scalar buffer surface, sign extend to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_i8/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_i8.json"},{"mnemonic":"s_buffer_load_u16","slug":"s_buffer_load_u16","records":1,"summary":"Load 16 bits of unsigned data from a scalar buffer surface, zero extend to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_u16/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_u16.json"},{"mnemonic":"s_buffer_load_u8","slug":"s_buffer_load_u8","records":1,"summary":"Load 8 bits of unsigned data from a scalar buffer surface, zero extend to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_buffer_load_u8/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_load_u8.json"},{"mnemonic":"s_buffer_prefetch_data","slug":"s_buffer_prefetch_data","records":1,"summary":"Prefetch data into the scalar data cache, relative to a base address provided in a resource descriptor constant.","page":"https://instructionsets.com/amdgpu/s_buffer_prefetch_data/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_prefetch_data.json"},{"mnemonic":"s_buffer_store_dword","slug":"s_buffer_store_dword","records":1,"summary":"Store 32 bits of data from a scalar register into a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_store_dword/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_store_dword.json"},{"mnemonic":"s_buffer_store_dwordx2","slug":"s_buffer_store_dwordx2","records":1,"summary":"Store 64 bits of data from a scalar register into a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_store_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_store_dwordx2.json"},{"mnemonic":"s_buffer_store_dwordx4","slug":"s_buffer_store_dwordx4","records":1,"summary":"Store 128 bits of data from a scalar register into a scalar buffer surface.","page":"https://instructionsets.com/amdgpu/s_buffer_store_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/s_buffer_store_dwordx4.json"},{"mnemonic":"s_call_b64","slug":"s_call_b64","records":1,"summary":"Store the address of the next instruction to a scalar register and then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_call_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_call_b64.json","aliases":["s_call_i64"]},{"mnemonic":"s_call_i64","slug":"s_call_i64","records":1,"summary":"AMDGPU SOPK scalar instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_call_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_call_i64.json","aliases":["s_call_b64"]},{"mnemonic":"s_cbranch_cdbgsys","slug":"s_cbranch_cdbgsys","records":1,"summary":"If the system debug flag is set then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_cbranch_cdbgsys/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_cdbgsys.json"},{"mnemonic":"s_cbranch_cdbgsys_and_user","slug":"s_cbranch_cdbgsys_and_user","records":1,"summary":"If the system and user debug flag are set then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_cbranch_cdbgsys_and_user/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_cdbgsys_and_user.json"},{"mnemonic":"s_cbranch_cdbgsys_or_user","slug":"s_cbranch_cdbgsys_or_user","records":1,"summary":"If the system or user debug flag is set then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_cbranch_cdbgsys_or_user/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_cdbgsys_or_user.json"},{"mnemonic":"s_cbranch_cdbguser","slug":"s_cbranch_cdbguser","records":1,"summary":"If the user debug flag is set then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_cbranch_cdbguser/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_cdbguser.json"},{"mnemonic":"s_cbranch_execnz","slug":"s_cbranch_execnz","records":1,"summary":"If EXECZ is 0 then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_cbranch_execnz/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_execnz.json"},{"mnemonic":"s_cbranch_execz","slug":"s_cbranch_execz","records":1,"summary":"If EXECZ is 1 then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_cbranch_execz/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_execz.json"},{"mnemonic":"s_cbranch_g_fork","slug":"s_cbranch_g_fork","records":1,"summary":"Conditional branch using branch-stack.","page":"https://instructionsets.com/amdgpu/s_cbranch_g_fork/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_g_fork.json"},{"mnemonic":"s_cbranch_i_fork","slug":"s_cbranch_i_fork","records":1,"summary":"Conditional branch using branch-stack.","page":"https://instructionsets.com/amdgpu/s_cbranch_i_fork/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_i_fork.json"},{"mnemonic":"s_cbranch_join","slug":"s_cbranch_join","records":1,"summary":"Conditional branch join point (end of conditional branch block).","page":"https://instructionsets.com/amdgpu/s_cbranch_join/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_join.json"},{"mnemonic":"s_cbranch_scc0","slug":"s_cbranch_scc0","records":1,"summary":"If SCC is 0 then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_cbranch_scc0/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_scc0.json"},{"mnemonic":"s_cbranch_scc1","slug":"s_cbranch_scc1","records":1,"summary":"Conditional relative branch, taken when the SCC (scalar condition code) flag is set.","page":"https://instructionsets.com/amdgpu/s_cbranch_scc1/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_scc1.json"},{"mnemonic":"s_cbranch_vccnz","slug":"s_cbranch_vccnz","records":1,"summary":"If VCCZ is 0 then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_cbranch_vccnz/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_vccnz.json"},{"mnemonic":"s_cbranch_vccz","slug":"s_cbranch_vccz","records":1,"summary":"If VCCZ is 1 then jump to a constant offset relative to the current PC.","page":"https://instructionsets.com/amdgpu/s_cbranch_vccz/","api":"https://instructionsets.com/api/v1/amdgpu/s_cbranch_vccz.json"},{"mnemonic":"s_ceil_f16","slug":"s_ceil_f16","records":1,"summary":"Round the half-precision float input up to next integer and store the result in floating point format into a scalar register.","page":"https://instructionsets.com/amdgpu/s_ceil_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_ceil_f16.json"},{"mnemonic":"s_ceil_f32","slug":"s_ceil_f32","records":1,"summary":"Round the single-precision float input up to next integer and store the result in floating point format into a scalar register.","page":"https://instructionsets.com/amdgpu/s_ceil_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_ceil_f32.json"},{"mnemonic":"s_clause","slug":"s_clause","records":1,"summary":"Mark the beginning of a clause.","page":"https://instructionsets.com/amdgpu/s_clause/","api":"https://instructionsets.com/api/v1/amdgpu/s_clause.json"},{"mnemonic":"s_cls_i32","slug":"s_cls_i32","records":1,"summary":"Count the number of leading bits that are the same as the sign bit of a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cls_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cls_i32.json","aliases":["s_flbit_i32"]},{"mnemonic":"s_cls_i32_i64","slug":"s_cls_i32_i64","records":1,"summary":"Count the number of leading bits that are the same as the sign bit of a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cls_i32_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_cls_i32_i64.json","aliases":["s_flbit_i32_i64"]},{"mnemonic":"s_clz_i32_u32","slug":"s_clz_i32_u32","records":1,"summary":"Count the number of leading \"0\" bits before the first \"1\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_clz_i32_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_clz_i32_u32.json","aliases":["s_flbit_i32_b32"]},{"mnemonic":"s_clz_i32_u64","slug":"s_clz_i32_u64","records":1,"summary":"Count the number of leading \"0\" bits before the first \"1\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_clz_i32_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_clz_i32_u64.json","aliases":["s_flbit_i32_b64"]},{"mnemonic":"s_cmov_b32","slug":"s_cmov_b32","records":1,"summary":"Move scalar input into a scalar register iff SCC is nonzero.","page":"https://instructionsets.com/amdgpu/s_cmov_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmov_b32.json"},{"mnemonic":"s_cmov_b64","slug":"s_cmov_b64","records":1,"summary":"Move scalar input into a scalar register iff SCC is nonzero.","page":"https://instructionsets.com/amdgpu/s_cmov_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmov_b64.json"},{"mnemonic":"s_cmovk_i32","slug":"s_cmovk_i32","records":1,"summary":"Move the sign extension of a literal 16-bit constant into a scalar register iff SCC is nonzero.","page":"https://instructionsets.com/amdgpu/s_cmovk_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmovk_i32.json"},{"mnemonic":"s_cmp_eq_f16","slug":"s_cmp_eq_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_eq_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_eq_f16.json"},{"mnemonic":"s_cmp_eq_f32","slug":"s_cmp_eq_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_eq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_eq_f32.json"},{"mnemonic":"s_cmp_eq_i32","slug":"s_cmp_eq_i32","records":1,"summary":"Scalar signed-32-bit equality compare, result written to SCC.","page":"https://instructionsets.com/amdgpu/s_cmp_eq_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_eq_i32.json"},{"mnemonic":"s_cmp_eq_u32","slug":"s_cmp_eq_u32","records":1,"summary":"Set SCC to 1 iff the first scalar input is equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_eq_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_eq_u32.json"},{"mnemonic":"s_cmp_eq_u64","slug":"s_cmp_eq_u64","records":1,"summary":"Set SCC to 1 iff the first scalar input is equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_eq_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_eq_u64.json"},{"mnemonic":"s_cmp_ge_f16","slug":"s_cmp_ge_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is greater than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_ge_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_ge_f16.json"},{"mnemonic":"s_cmp_ge_f32","slug":"s_cmp_ge_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is greater than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_ge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_ge_f32.json"},{"mnemonic":"s_cmp_ge_i32","slug":"s_cmp_ge_i32","records":1,"summary":"Set SCC to 1 iff the first scalar input is greater than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_ge_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_ge_i32.json"},{"mnemonic":"s_cmp_ge_u32","slug":"s_cmp_ge_u32","records":1,"summary":"Set SCC to 1 iff the first scalar input is greater than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_ge_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_ge_u32.json"},{"mnemonic":"s_cmp_gt_f16","slug":"s_cmp_gt_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_gt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_gt_f16.json"},{"mnemonic":"s_cmp_gt_f32","slug":"s_cmp_gt_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_gt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_gt_f32.json"},{"mnemonic":"s_cmp_gt_i32","slug":"s_cmp_gt_i32","records":1,"summary":"Set SCC to 1 iff the first scalar input is greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_gt_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_gt_i32.json"},{"mnemonic":"s_cmp_gt_u32","slug":"s_cmp_gt_u32","records":1,"summary":"Set SCC to 1 iff the first scalar input is greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_gt_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_gt_u32.json"},{"mnemonic":"s_cmp_le_f16","slug":"s_cmp_le_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_le_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_le_f16.json"},{"mnemonic":"s_cmp_le_f32","slug":"s_cmp_le_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_le_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_le_f32.json"},{"mnemonic":"s_cmp_le_i32","slug":"s_cmp_le_i32","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_le_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_le_i32.json"},{"mnemonic":"s_cmp_le_u32","slug":"s_cmp_le_u32","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_le_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_le_u32.json"},{"mnemonic":"s_cmp_lg_f16","slug":"s_cmp_lg_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than or greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_lg_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_lg_f16.json"},{"mnemonic":"s_cmp_lg_f32","slug":"s_cmp_lg_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than or greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_lg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_lg_f32.json"},{"mnemonic":"s_cmp_lg_i32","slug":"s_cmp_lg_i32","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than or greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_lg_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_lg_i32.json"},{"mnemonic":"s_cmp_lg_u32","slug":"s_cmp_lg_u32","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than or greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_lg_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_lg_u32.json"},{"mnemonic":"s_cmp_lg_u64","slug":"s_cmp_lg_u64","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than or greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_lg_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_lg_u64.json"},{"mnemonic":"s_cmp_lt_f16","slug":"s_cmp_lt_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_lt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_lt_f16.json"},{"mnemonic":"s_cmp_lt_f32","slug":"s_cmp_lt_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_lt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_lt_f32.json"},{"mnemonic":"s_cmp_lt_i32","slug":"s_cmp_lt_i32","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_lt_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_lt_i32.json"},{"mnemonic":"s_cmp_lt_u32","slug":"s_cmp_lt_u32","records":1,"summary":"Set SCC to 1 iff the first scalar input is less than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_lt_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_lt_u32.json"},{"mnemonic":"s_cmp_neq_f16","slug":"s_cmp_neq_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is not equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_neq_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_neq_f16.json"},{"mnemonic":"s_cmp_neq_f32","slug":"s_cmp_neq_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is not equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_neq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_neq_f32.json"},{"mnemonic":"s_cmp_nge_f16","slug":"s_cmp_nge_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is not greater than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_nge_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_nge_f16.json"},{"mnemonic":"s_cmp_nge_f32","slug":"s_cmp_nge_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is not greater than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_nge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_nge_f32.json"},{"mnemonic":"s_cmp_ngt_f16","slug":"s_cmp_ngt_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is not greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_ngt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_ngt_f16.json"},{"mnemonic":"s_cmp_ngt_f32","slug":"s_cmp_ngt_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is not greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_ngt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_ngt_f32.json"},{"mnemonic":"s_cmp_nle_f16","slug":"s_cmp_nle_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is not less than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_nle_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_nle_f16.json"},{"mnemonic":"s_cmp_nle_f32","slug":"s_cmp_nle_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is not less than or equal to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_nle_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_nle_f32.json"},{"mnemonic":"s_cmp_nlg_f16","slug":"s_cmp_nlg_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is not less than or greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_nlg_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_nlg_f16.json"},{"mnemonic":"s_cmp_nlg_f32","slug":"s_cmp_nlg_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is not less than or greater than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_nlg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_nlg_f32.json"},{"mnemonic":"s_cmp_nlt_f16","slug":"s_cmp_nlt_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is not less than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_nlt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_nlt_f16.json"},{"mnemonic":"s_cmp_nlt_f32","slug":"s_cmp_nlt_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is not less than the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_nlt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_nlt_f32.json"},{"mnemonic":"s_cmp_o_f16","slug":"s_cmp_o_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is orderable to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_o_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_o_f16.json"},{"mnemonic":"s_cmp_o_f32","slug":"s_cmp_o_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is orderable to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_o_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_o_f32.json"},{"mnemonic":"s_cmp_u_f16","slug":"s_cmp_u_f16","records":1,"summary":"Set SCC to 1 iff the first scalar input is not orderable to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_u_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_u_f16.json"},{"mnemonic":"s_cmp_u_f32","slug":"s_cmp_u_f32","records":1,"summary":"Set SCC to 1 iff the first scalar input is not orderable to the second scalar input.","page":"https://instructionsets.com/amdgpu/s_cmp_u_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmp_u_f32.json"},{"mnemonic":"s_cmpk_eq_i32","slug":"s_cmpk_eq_i32","records":1,"summary":"Set SCC to 1 iff scalar input is equal to the sign extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_eq_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_eq_i32.json"},{"mnemonic":"s_cmpk_eq_u32","slug":"s_cmpk_eq_u32","records":1,"summary":"Set SCC to 1 iff scalar input is equal to the zero extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_eq_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_eq_u32.json"},{"mnemonic":"s_cmpk_ge_i32","slug":"s_cmpk_ge_i32","records":1,"summary":"Set SCC to 1 iff scalar input is greater than or equal to the sign extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_ge_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_ge_i32.json"},{"mnemonic":"s_cmpk_ge_u32","slug":"s_cmpk_ge_u32","records":1,"summary":"Set SCC to 1 iff scalar input is greater than or equal to the zero extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_ge_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_ge_u32.json"},{"mnemonic":"s_cmpk_gt_i32","slug":"s_cmpk_gt_i32","records":1,"summary":"Set SCC to 1 iff scalar input is greater than the sign extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_gt_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_gt_i32.json"},{"mnemonic":"s_cmpk_gt_u32","slug":"s_cmpk_gt_u32","records":1,"summary":"Set SCC to 1 iff scalar input is greater than the zero extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_gt_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_gt_u32.json"},{"mnemonic":"s_cmpk_le_i32","slug":"s_cmpk_le_i32","records":1,"summary":"Set SCC to 1 iff scalar input is less than or equal to the sign extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_le_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_le_i32.json"},{"mnemonic":"s_cmpk_le_u32","slug":"s_cmpk_le_u32","records":1,"summary":"Set SCC to 1 iff scalar input is less than or equal to the zero extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_le_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_le_u32.json"},{"mnemonic":"s_cmpk_lg_i32","slug":"s_cmpk_lg_i32","records":1,"summary":"Set SCC to 1 iff scalar input is less than or greater than the sign extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_lg_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_lg_i32.json"},{"mnemonic":"s_cmpk_lg_u32","slug":"s_cmpk_lg_u32","records":1,"summary":"Set SCC to 1 iff scalar input is less than or greater than the zero extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_lg_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_lg_u32.json"},{"mnemonic":"s_cmpk_lt_i32","slug":"s_cmpk_lt_i32","records":1,"summary":"Set SCC to 1 iff scalar input is less than the sign extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_lt_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_lt_i32.json"},{"mnemonic":"s_cmpk_lt_u32","slug":"s_cmpk_lt_u32","records":1,"summary":"Set SCC to 1 iff scalar input is less than the zero extension of a literal 16-bit constant.","page":"https://instructionsets.com/amdgpu/s_cmpk_lt_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cmpk_lt_u32.json"},{"mnemonic":"s_code_end","slug":"s_code_end","records":1,"summary":"Generate an illegal instruction interrupt. This instruction is used to mark the end of a shader buffer for debug tools.","page":"https://instructionsets.com/amdgpu/s_code_end/","api":"https://instructionsets.com/api/v1/amdgpu/s_code_end.json"},{"mnemonic":"s_cselect_b32","slug":"s_cselect_b32","records":1,"summary":"Select the first input if SCC is true otherwise select the second input, then store the selected input into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cselect_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cselect_b32.json"},{"mnemonic":"s_cselect_b64","slug":"s_cselect_b64","records":1,"summary":"Select the first input if SCC is true otherwise select the second input, then store the selected input into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cselect_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_cselect_b64.json"},{"mnemonic":"s_ctz_i32_b32","slug":"s_ctz_i32_b32","records":1,"summary":"Count the number of trailing \"0\" bits before the first \"1\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_ctz_i32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_ctz_i32_b32.json","aliases":["s_ff1_i32_b32"]},{"mnemonic":"s_ctz_i32_b64","slug":"s_ctz_i32_b64","records":1,"summary":"Count the number of trailing \"0\" bits before the first \"1\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_ctz_i32_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_ctz_i32_b64.json","aliases":["s_ff1_i32_b64"]},{"mnemonic":"s_cvt_f16_f32","slug":"s_cvt_f16_f32","records":1,"summary":"Convert from a single-precision float input to a half-precision float value and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cvt_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cvt_f16_f32.json"},{"mnemonic":"s_cvt_f32_f16","slug":"s_cvt_f32_f16","records":1,"summary":"Convert from a half-precision float input to a single-precision float value and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cvt_f32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cvt_f32_f16.json"},{"mnemonic":"s_cvt_f32_i32","slug":"s_cvt_f32_i32","records":1,"summary":"Convert from a signed 32-bit integer input to a single-precision float value and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cvt_f32_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cvt_f32_i32.json"},{"mnemonic":"s_cvt_f32_u32","slug":"s_cvt_f32_u32","records":1,"summary":"Convert from an unsigned 32-bit integer input to a single-precision float value and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cvt_f32_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cvt_f32_u32.json"},{"mnemonic":"s_cvt_hi_f32_f16","slug":"s_cvt_hi_f32_f16","records":1,"summary":"Convert from a half-precision float value in the high 16 bits of a scalar input to a single-precision float value and store the result into a scalar…","page":"https://instructionsets.com/amdgpu/s_cvt_hi_f32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_cvt_hi_f32_f16.json"},{"mnemonic":"s_cvt_i32_f32","slug":"s_cvt_i32_f32","records":1,"summary":"Convert from a single-precision float input to a signed 32-bit integer value and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cvt_i32_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cvt_i32_f32.json"},{"mnemonic":"s_cvt_pk_rtz_f16_f32","slug":"s_cvt_pk_rtz_f16_f32","records":1,"summary":"Convert two single-precision float inputs into a packed half-precision float result using round toward zero semantics (ignore the current rounding…","page":"https://instructionsets.com/amdgpu/s_cvt_pk_rtz_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cvt_pk_rtz_f16_f32.json"},{"mnemonic":"s_cvt_u32_f32","slug":"s_cvt_u32_f32","records":1,"summary":"Convert from a single-precision float input to an unsigned 32-bit integer value and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_cvt_u32_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_cvt_u32_f32.json"},{"mnemonic":"s_dcache_discard","slug":"s_dcache_discard","records":1,"summary":"Discard one dirty scalar data L0 cache line. A cache line is 64 bytes.","page":"https://instructionsets.com/amdgpu/s_dcache_discard/","api":"https://instructionsets.com/api/v1/amdgpu/s_dcache_discard.json"},{"mnemonic":"s_dcache_discard_x2","slug":"s_dcache_discard_x2","records":1,"summary":"Discard two consecutive dirty scalar data L0 cache lines. A cache line is 64 bytes.","page":"https://instructionsets.com/amdgpu/s_dcache_discard_x2/","api":"https://instructionsets.com/api/v1/amdgpu/s_dcache_discard_x2.json"},{"mnemonic":"s_dcache_inv","slug":"s_dcache_inv","records":1,"summary":"Invalidate the scalar (L0) data cache.","page":"https://instructionsets.com/amdgpu/s_dcache_inv/","api":"https://instructionsets.com/api/v1/amdgpu/s_dcache_inv.json"},{"mnemonic":"s_dcache_inv_vol","slug":"s_dcache_inv_vol","records":1,"summary":"Invalidate the scalar (L0) data cache volatile lines.","page":"https://instructionsets.com/amdgpu/s_dcache_inv_vol/","api":"https://instructionsets.com/api/v1/amdgpu/s_dcache_inv_vol.json"},{"mnemonic":"s_dcache_wb","slug":"s_dcache_wb","records":1,"summary":"Write back dirty data in the scalar (L0) data cache.","page":"https://instructionsets.com/amdgpu/s_dcache_wb/","api":"https://instructionsets.com/api/v1/amdgpu/s_dcache_wb.json"},{"mnemonic":"s_dcache_wb_vol","slug":"s_dcache_wb_vol","records":1,"summary":"Write back dirty data in the scalar (L0) data cache volatile lines.","page":"https://instructionsets.com/amdgpu/s_dcache_wb_vol/","api":"https://instructionsets.com/api/v1/amdgpu/s_dcache_wb_vol.json"},{"mnemonic":"s_decperflevel","slug":"s_decperflevel","records":1,"summary":"Decrement performance counter specified in SIMM16[3:0] by 1.","page":"https://instructionsets.com/amdgpu/s_decperflevel/","api":"https://instructionsets.com/api/v1/amdgpu/s_decperflevel.json"},{"mnemonic":"s_delay_alu","slug":"s_delay_alu","records":1,"summary":"Insert delay between dependent SALU/VALU instructions.","page":"https://instructionsets.com/amdgpu/s_delay_alu/","api":"https://instructionsets.com/api/v1/amdgpu/s_delay_alu.json"},{"mnemonic":"s_denorm_mode","slug":"s_denorm_mode","records":1,"summary":"Set floating point denormal mode using an immediate constant.","page":"https://instructionsets.com/amdgpu/s_denorm_mode/","api":"https://instructionsets.com/api/v1/amdgpu/s_denorm_mode.json"},{"mnemonic":"s_endpgm","slug":"s_endpgm","records":1,"summary":"End of program; terminate wavefront.","page":"https://instructionsets.com/amdgpu/s_endpgm/","api":"https://instructionsets.com/api/v1/amdgpu/s_endpgm.json"},{"mnemonic":"s_endpgm_ordered_ps_done","slug":"s_endpgm_ordered_ps_done","records":1,"summary":"End of program; signal that a wave has exited its POPS critical section and terminate wavefront.","page":"https://instructionsets.com/amdgpu/s_endpgm_ordered_ps_done/","api":"https://instructionsets.com/api/v1/amdgpu/s_endpgm_ordered_ps_done.json"},{"mnemonic":"s_endpgm_saved","slug":"s_endpgm_saved","records":1,"summary":"End of program; signal that a wave has been saved by the context-switch trap handler and terminate wavefront.","page":"https://instructionsets.com/amdgpu/s_endpgm_saved/","api":"https://instructionsets.com/api/v1/amdgpu/s_endpgm_saved.json"},{"mnemonic":"s_ff0_i32_b32","slug":"s_ff0_i32_b32","records":1,"summary":"Count the number of trailing \"1\" bits before the first \"0\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_ff0_i32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_ff0_i32_b32.json"},{"mnemonic":"s_ff0_i32_b64","slug":"s_ff0_i32_b64","records":1,"summary":"Count the number of trailing \"1\" bits before the first \"0\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_ff0_i32_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_ff0_i32_b64.json"},{"mnemonic":"s_ff1_i32_b32","slug":"s_ff1_i32_b32","records":1,"summary":"Count the number of trailing \"0\" bits before the first \"1\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_ff1_i32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_ff1_i32_b32.json","aliases":["s_ctz_i32_b32"]},{"mnemonic":"s_ff1_i32_b64","slug":"s_ff1_i32_b64","records":1,"summary":"Count the number of trailing \"0\" bits before the first \"1\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_ff1_i32_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_ff1_i32_b64.json","aliases":["s_ctz_i32_b64"]},{"mnemonic":"s_flbit_i32","slug":"s_flbit_i32","records":1,"summary":"Count the number of leading bits that are the same as the sign bit of a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_flbit_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_flbit_i32.json","aliases":["s_cls_i32"]},{"mnemonic":"s_flbit_i32_b32","slug":"s_flbit_i32_b32","records":1,"summary":"Count the number of leading \"0\" bits before the first \"1\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_flbit_i32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_flbit_i32_b32.json","aliases":["s_clz_i32_u32"]},{"mnemonic":"s_flbit_i32_b64","slug":"s_flbit_i32_b64","records":1,"summary":"Count the number of leading \"0\" bits before the first \"1\" in a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_flbit_i32_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_flbit_i32_b64.json","aliases":["s_clz_i32_u64"]},{"mnemonic":"s_flbit_i32_i64","slug":"s_flbit_i32_i64","records":1,"summary":"Count the number of leading bits that are the same as the sign bit of a scalar input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_flbit_i32_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_flbit_i32_i64.json","aliases":["s_cls_i32_i64"]},{"mnemonic":"s_floor_f16","slug":"s_floor_f16","records":1,"summary":"Round the half-precision float input down to previous integer and store the result in floating point format into a scalar register.","page":"https://instructionsets.com/amdgpu/s_floor_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_floor_f16.json"},{"mnemonic":"s_floor_f32","slug":"s_floor_f32","records":1,"summary":"Round the single-precision float input down to previous integer and store the result in floating point format into a scalar register.","page":"https://instructionsets.com/amdgpu/s_floor_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_floor_f32.json"},{"mnemonic":"s_fmaak_f32","slug":"s_fmaak_f32","records":1,"summary":"Multiply two single-precision float inputs and add a literal constant using fused multiply add, and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_fmaak_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_fmaak_f32.json"},{"mnemonic":"s_fmac_f16","slug":"s_fmac_f16","records":1,"summary":"Multiply two half-precision float inputs and accumulate the result into the destination register using fused multiply add.","page":"https://instructionsets.com/amdgpu/s_fmac_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_fmac_f16.json"},{"mnemonic":"s_fmac_f32","slug":"s_fmac_f32","records":1,"summary":"Multiply two single-precision float inputs and accumulate the result into the destination register using fused multiply add.","page":"https://instructionsets.com/amdgpu/s_fmac_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_fmac_f32.json"},{"mnemonic":"s_fmamk_f32","slug":"s_fmamk_f32","records":1,"summary":"Multiply a single-precision float input with a literal constant and add a second single-precision float input using fused multiply add, and store the…","page":"https://instructionsets.com/amdgpu/s_fmamk_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_fmamk_f32.json"},{"mnemonic":"s_get_barrier_state","slug":"s_get_barrier_state","records":1,"summary":"AMDGPU SOP1 scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_get_barrier_state/","api":"https://instructionsets.com/api/v1/amdgpu/s_get_barrier_state.json"},{"mnemonic":"s_get_pc_i64","slug":"s_get_pc_i64","records":1,"summary":"AMDGPU SOP1 scalar instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_get_pc_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_get_pc_i64.json","aliases":["s_getpc_b64"]},{"mnemonic":"s_get_shader_cycles_u64","slug":"s_get_shader_cycles_u64","records":1,"summary":"AMDGPU SOP1 scalar instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_get_shader_cycles_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_get_shader_cycles_u64.json"},{"mnemonic":"s_get_waveid_in_workgroup","slug":"s_get_waveid_in_workgroup","records":1,"summary":"Return the wave's ID within a workgroup 0-(N-1).","page":"https://instructionsets.com/amdgpu/s_get_waveid_in_workgroup/","api":"https://instructionsets.com/api/v1/amdgpu/s_get_waveid_in_workgroup.json"},{"mnemonic":"s_getpc_b64","slug":"s_getpc_b64","records":1,"summary":"Store the address of the next instruction to a scalar register.","page":"https://instructionsets.com/amdgpu/s_getpc_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_getpc_b64.json","aliases":["s_get_pc_i64"]},{"mnemonic":"s_getreg_b32","slug":"s_getreg_b32","records":1,"summary":"Read some or all of a hardware register into the LSBs of destination.","page":"https://instructionsets.com/amdgpu/s_getreg_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_getreg_b32.json"},{"mnemonic":"s_gl1_inv","slug":"s_gl1_inv","records":1,"summary":"Invalidate the GL1 cache only.","page":"https://instructionsets.com/amdgpu/s_gl1_inv/","api":"https://instructionsets.com/api/v1/amdgpu/s_gl1_inv.json"},{"mnemonic":"s_icache_inv","slug":"s_icache_inv","records":1,"summary":"Invalidate entire first level instruction cache.","page":"https://instructionsets.com/amdgpu/s_icache_inv/","api":"https://instructionsets.com/api/v1/amdgpu/s_icache_inv.json"},{"mnemonic":"s_incperflevel","slug":"s_incperflevel","records":1,"summary":"Increment performance counter specified in SIMM16[3:0] by 1.","page":"https://instructionsets.com/amdgpu/s_incperflevel/","api":"https://instructionsets.com/api/v1/amdgpu/s_incperflevel.json"},{"mnemonic":"s_inst_prefetch","slug":"s_inst_prefetch","records":1,"summary":"Change instruction prefetch mode. This controls how many cachelines ahead of the current PC the shader attempts to prefetch.","page":"https://instructionsets.com/amdgpu/s_inst_prefetch/","api":"https://instructionsets.com/api/v1/amdgpu/s_inst_prefetch.json","aliases":["s_set_inst_prefetch_distance"]},{"mnemonic":"s_load_b128","slug":"s_load_b128","records":1,"summary":"Load 128 bits of data from the scalar memory into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_b128/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_b128.json","aliases":["s_load_dwordx4"]},{"mnemonic":"s_load_b256","slug":"s_load_b256","records":1,"summary":"Load 256 bits of data from the scalar memory into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_b256/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_b256.json","aliases":["s_load_dwordx8"]},{"mnemonic":"s_load_b512","slug":"s_load_b512","records":1,"summary":"Load 512 bits of data from the scalar memory into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_b512/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_b512.json","aliases":["s_load_dwordx16"]},{"mnemonic":"s_load_b64","slug":"s_load_b64","records":1,"summary":"Load 64 bits of data from the scalar memory into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_b64.json","aliases":["s_load_dwordx2"]},{"mnemonic":"s_load_b96","slug":"s_load_b96","records":1,"summary":"Load 96 bits of data from the scalar memory into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_b96/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_b96.json"},{"mnemonic":"s_load_dword","slug":"s_load_dword","records":1,"summary":"Load one 32-bit dword from memory into a scalar register, wavefront-uniform.","page":"https://instructionsets.com/amdgpu/s_load_dword/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_dword.json","aliases":["s_load_b32"]},{"mnemonic":"s_load_dwordx16","slug":"s_load_dwordx16","records":1,"summary":"Load 512 bits of data from the scalar memory into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_dwordx16/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_dwordx16.json","aliases":["s_load_b512"]},{"mnemonic":"s_load_dwordx2","slug":"s_load_dwordx2","records":1,"summary":"Load 64 bits of data from the scalar memory into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_dwordx2.json","aliases":["s_load_b64"]},{"mnemonic":"s_load_dwordx4","slug":"s_load_dwordx4","records":1,"summary":"Load 128 bits of data from the scalar memory into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_dwordx4.json","aliases":["s_load_b128"]},{"mnemonic":"s_load_dwordx8","slug":"s_load_dwordx8","records":1,"summary":"Load 256 bits of data from the scalar memory into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_dwordx8/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_dwordx8.json","aliases":["s_load_b256"]},{"mnemonic":"s_load_i16","slug":"s_load_i16","records":1,"summary":"Load 16 bits of signed data from the scalar memory, sign extend to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_i16/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_i16.json"},{"mnemonic":"s_load_i8","slug":"s_load_i8","records":1,"summary":"Load 8 bits of signed data from the scalar memory, sign extend to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_i8/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_i8.json"},{"mnemonic":"s_load_u16","slug":"s_load_u16","records":1,"summary":"Load 16 bits of unsigned data from the scalar memory, zero extend to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_u16/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_u16.json"},{"mnemonic":"s_load_u8","slug":"s_load_u8","records":1,"summary":"Load 8 bits of unsigned data from the scalar memory, zero extend to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_load_u8/","api":"https://instructionsets.com/api/v1/amdgpu/s_load_u8.json"},{"mnemonic":"s_lshl1_add_u32","slug":"s_lshl1_add_u32","records":1,"summary":"Calculate the logical shift left of the first input by 1, then add the second input, store the result into a scalar register and set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_lshl1_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_lshl1_add_u32.json"},{"mnemonic":"s_lshl2_add_u32","slug":"s_lshl2_add_u32","records":1,"summary":"Calculate the logical shift left of the first input by 2, then add the second input, store the result into a scalar register and set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_lshl2_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_lshl2_add_u32.json"},{"mnemonic":"s_lshl3_add_u32","slug":"s_lshl3_add_u32","records":1,"summary":"Calculate the logical shift left of the first input by 3, then add the second input, store the result into a scalar register and set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_lshl3_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_lshl3_add_u32.json"},{"mnemonic":"s_lshl4_add_u32","slug":"s_lshl4_add_u32","records":1,"summary":"Calculate the logical shift left of the first input by 4, then add the second input, store the result into a scalar register and set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_lshl4_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_lshl4_add_u32.json"},{"mnemonic":"s_lshl_b32","slug":"s_lshl_b32","records":1,"summary":"Given a shift count in the second scalar input, calculate the logical shift left of the first scalar input, store the result into a scalar register…","page":"https://instructionsets.com/amdgpu/s_lshl_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_lshl_b32.json"},{"mnemonic":"s_lshl_b64","slug":"s_lshl_b64","records":1,"summary":"Given a shift count in the second scalar input, calculate the logical shift left of the first scalar input, store the result into a scalar register…","page":"https://instructionsets.com/amdgpu/s_lshl_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_lshl_b64.json"},{"mnemonic":"s_lshr_b32","slug":"s_lshr_b32","records":1,"summary":"Given a shift count in the second scalar input, calculate the logical shift right of the first scalar input, store the result into a scalar register…","page":"https://instructionsets.com/amdgpu/s_lshr_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_lshr_b32.json"},{"mnemonic":"s_lshr_b64","slug":"s_lshr_b64","records":1,"summary":"Given a shift count in the second scalar input, calculate the logical shift right of the first scalar input, store the result into a scalar register…","page":"https://instructionsets.com/amdgpu/s_lshr_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_lshr_b64.json"},{"mnemonic":"s_max_f16","slug":"s_max_f16","records":1,"summary":"Select the maximum of two half-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_max_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_max_f16.json","aliases":["s_max_num_f16"]},{"mnemonic":"s_max_f32","slug":"s_max_f32","records":1,"summary":"Select the maximum of two single-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_max_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_max_f32.json","aliases":["s_max_num_f32"]},{"mnemonic":"s_max_i32","slug":"s_max_i32","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs, store the selected value into a scalar register and set SCC iff the first value is selected.","page":"https://instructionsets.com/amdgpu/s_max_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_max_i32.json"},{"mnemonic":"s_max_num_f16","slug":"s_max_num_f16","records":1,"summary":"Select the IEEE maximumNumber() of two half-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_max_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_max_num_f16.json","aliases":["s_max_f16"]},{"mnemonic":"s_max_num_f32","slug":"s_max_num_f32","records":1,"summary":"Select the IEEE maximumNumber() of two single-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_max_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_max_num_f32.json","aliases":["s_max_f32"]},{"mnemonic":"s_max_u32","slug":"s_max_u32","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs, store the selected value into a scalar register and set SCC iff the first value is selected.","page":"https://instructionsets.com/amdgpu/s_max_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_max_u32.json"},{"mnemonic":"s_maximum_f16","slug":"s_maximum_f16","records":1,"summary":"Select the IEEE maximum() of two half-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_maximum_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_maximum_f16.json"},{"mnemonic":"s_maximum_f32","slug":"s_maximum_f32","records":1,"summary":"Select the IEEE maximum() of two single-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_maximum_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_maximum_f32.json"},{"mnemonic":"s_memrealtime","slug":"s_memrealtime","records":1,"summary":"Return current 64-bit RTC.","page":"https://instructionsets.com/amdgpu/s_memrealtime/","api":"https://instructionsets.com/api/v1/amdgpu/s_memrealtime.json"},{"mnemonic":"s_memtime","slug":"s_memtime","records":1,"summary":"Return current 64-bit timestamp.","page":"https://instructionsets.com/amdgpu/s_memtime/","api":"https://instructionsets.com/api/v1/amdgpu/s_memtime.json"},{"mnemonic":"s_min_f16","slug":"s_min_f16","records":1,"summary":"Select the minimum of two half-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_min_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_min_f16.json","aliases":["s_min_num_f16"]},{"mnemonic":"s_min_f32","slug":"s_min_f32","records":1,"summary":"Select the minimum of two single-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_min_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_min_f32.json","aliases":["s_min_num_f32"]},{"mnemonic":"s_min_i32","slug":"s_min_i32","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs, store the selected value into a scalar register and set SCC iff the first value is selected.","page":"https://instructionsets.com/amdgpu/s_min_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_min_i32.json"},{"mnemonic":"s_min_num_f16","slug":"s_min_num_f16","records":1,"summary":"Select the IEEE minimumNumber() of two half-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_min_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_min_num_f16.json","aliases":["s_min_f16"]},{"mnemonic":"s_min_num_f32","slug":"s_min_num_f32","records":1,"summary":"Select the IEEE minimumNumber() of two single-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_min_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_min_num_f32.json","aliases":["s_min_f32"]},{"mnemonic":"s_min_u32","slug":"s_min_u32","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs, store the selected value into a scalar register and set SCC iff the first value is selected.","page":"https://instructionsets.com/amdgpu/s_min_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_min_u32.json"},{"mnemonic":"s_minimum_f16","slug":"s_minimum_f16","records":1,"summary":"Select the IEEE minimum() of two half-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_minimum_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_minimum_f16.json"},{"mnemonic":"s_minimum_f32","slug":"s_minimum_f32","records":1,"summary":"Select the IEEE minimum() of two single-precision float inputs and store the selected value into a scalar register.","page":"https://instructionsets.com/amdgpu/s_minimum_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_minimum_f32.json"},{"mnemonic":"s_monitor_sleep","slug":"s_monitor_sleep","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_monitor_sleep/","api":"https://instructionsets.com/api/v1/amdgpu/s_monitor_sleep.json"},{"mnemonic":"s_mov_b32","slug":"s_mov_b32","records":1,"summary":"Move scalar input into a scalar register.","page":"https://instructionsets.com/amdgpu/s_mov_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_mov_b32.json"},{"mnemonic":"s_mov_b64","slug":"s_mov_b64","records":1,"summary":"Move scalar input into a scalar register.","page":"https://instructionsets.com/amdgpu/s_mov_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_mov_b64.json"},{"mnemonic":"s_movk_i32","slug":"s_movk_i32","records":1,"summary":"Sign extend a literal 16-bit constant and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_movk_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_movk_i32.json"},{"mnemonic":"s_movreld_b32","slug":"s_movreld_b32","records":1,"summary":"Move data from a scalar input into a relatively-indexed scalar register.","page":"https://instructionsets.com/amdgpu/s_movreld_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_movreld_b32.json"},{"mnemonic":"s_movreld_b64","slug":"s_movreld_b64","records":1,"summary":"Move data from a scalar input into a relatively-indexed scalar register.","page":"https://instructionsets.com/amdgpu/s_movreld_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_movreld_b64.json"},{"mnemonic":"s_movrels_b32","slug":"s_movrels_b32","records":1,"summary":"Move data from a relatively-indexed scalar register into another scalar register.","page":"https://instructionsets.com/amdgpu/s_movrels_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_movrels_b32.json"},{"mnemonic":"s_movrels_b64","slug":"s_movrels_b64","records":1,"summary":"Move data from a relatively-indexed scalar register into another scalar register.","page":"https://instructionsets.com/amdgpu/s_movrels_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_movrels_b64.json"},{"mnemonic":"s_movrelsd_2_b32","slug":"s_movrelsd_2_b32","records":1,"summary":"Move data from a relatively-indexed scalar register into another relatively-indexed scalar register, using different offsets for each index.","page":"https://instructionsets.com/amdgpu/s_movrelsd_2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_movrelsd_2_b32.json"},{"mnemonic":"s_mul_f16","slug":"s_mul_f16","records":1,"summary":"Multiply two floating point inputs and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_mul_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_mul_f16.json"},{"mnemonic":"s_mul_f32","slug":"s_mul_f32","records":1,"summary":"Multiply two floating point inputs and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_mul_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_mul_f32.json"},{"mnemonic":"s_mul_hi_i32","slug":"s_mul_hi_i32","records":1,"summary":"Multiply two signed integers and store the high 32 bits of the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_mul_hi_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_mul_hi_i32.json"},{"mnemonic":"s_mul_hi_u32","slug":"s_mul_hi_u32","records":1,"summary":"Multiply two unsigned integers and store the high 32 bits of the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_mul_hi_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_mul_hi_u32.json"},{"mnemonic":"s_mul_i32","slug":"s_mul_i32","records":1,"summary":"Multiply two 32-bit signed scalar operands, wavefront-uniform, low 32 bits of the product.","page":"https://instructionsets.com/amdgpu/s_mul_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_mul_i32.json"},{"mnemonic":"s_mul_u64","slug":"s_mul_u64","records":1,"summary":"Multiply two unsigned 64-bit integer inputs and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_mul_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_mul_u64.json"},{"mnemonic":"s_mulk_i32","slug":"s_mulk_i32","records":1,"summary":"Multiply a scalar input with the sign extension of a literal 16-bit constant and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_mulk_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_mulk_i32.json"},{"mnemonic":"s_nand_b32","slug":"s_nand_b32","records":1,"summary":"Calculate bitwise NAND on two scalar inputs, store the result into a scalar register and set SCC if the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_nand_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_nand_b32.json"},{"mnemonic":"s_nand_b64","slug":"s_nand_b64","records":1,"summary":"Calculate bitwise NAND on two scalar inputs, store the result into a scalar register and set SCC if the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_nand_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_nand_b64.json"},{"mnemonic":"s_nand_saveexec_b32","slug":"s_nand_saveexec_b32","records":1,"summary":"Calculate bitwise NAND on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_nand_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_nand_saveexec_b32.json"},{"mnemonic":"s_nand_saveexec_b64","slug":"s_nand_saveexec_b64","records":1,"summary":"Calculate bitwise NAND on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_nand_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_nand_saveexec_b64.json"},{"mnemonic":"s_nop","slug":"s_nop","records":1,"summary":"Do nothing.","page":"https://instructionsets.com/amdgpu/s_nop/","api":"https://instructionsets.com/api/v1/amdgpu/s_nop.json"},{"mnemonic":"s_nor_b32","slug":"s_nor_b32","records":1,"summary":"Calculate bitwise NOR on two scalar inputs, store the result into a scalar register and set SCC if the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_nor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_nor_b32.json"},{"mnemonic":"s_nor_b64","slug":"s_nor_b64","records":1,"summary":"Calculate bitwise NOR on two scalar inputs, store the result into a scalar register and set SCC if the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_nor_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_nor_b64.json"},{"mnemonic":"s_nor_saveexec_b32","slug":"s_nor_saveexec_b32","records":1,"summary":"Calculate bitwise NOR on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_nor_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_nor_saveexec_b32.json"},{"mnemonic":"s_nor_saveexec_b64","slug":"s_nor_saveexec_b64","records":1,"summary":"Calculate bitwise NOR on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_nor_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_nor_saveexec_b64.json"},{"mnemonic":"s_not_b32","slug":"s_not_b32","records":1,"summary":"Calculate bitwise negation on a scalar input, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_not_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_not_b32.json"},{"mnemonic":"s_not_b64","slug":"s_not_b64","records":1,"summary":"Calculate bitwise negation on a scalar input, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_not_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_not_b64.json"},{"mnemonic":"s_or_b32","slug":"s_or_b32","records":1,"summary":"Calculate bitwise OR on two scalar inputs, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_or_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_b32.json"},{"mnemonic":"s_or_b64","slug":"s_or_b64","records":1,"summary":"Calculate bitwise OR on two scalar inputs, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_or_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_b64.json"},{"mnemonic":"s_or_not0_saveexec_b32","slug":"s_or_not0_saveexec_b32","records":1,"summary":"Calculate bitwise OR on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_or_not0_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_not0_saveexec_b32.json","aliases":["s_orn1_saveexec_b32"]},{"mnemonic":"s_or_not0_saveexec_b64","slug":"s_or_not0_saveexec_b64","records":1,"summary":"Calculate bitwise OR on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_or_not0_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_not0_saveexec_b64.json","aliases":["s_orn1_saveexec_b64"]},{"mnemonic":"s_or_not1_b32","slug":"s_or_not1_b32","records":1,"summary":"Calculate bitwise OR with the first input and the negation of the second input, store the result into a scalar register and set SCC if the result is…","page":"https://instructionsets.com/amdgpu/s_or_not1_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_not1_b32.json","aliases":["s_orn2_b32"]},{"mnemonic":"s_or_not1_b64","slug":"s_or_not1_b64","records":1,"summary":"Calculate bitwise OR with the first input and the negation of the second input, store the result into a scalar register and set SCC if the result is…","page":"https://instructionsets.com/amdgpu/s_or_not1_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_not1_b64.json","aliases":["s_orn2_b64"]},{"mnemonic":"s_or_not1_saveexec_b32","slug":"s_or_not1_saveexec_b32","records":1,"summary":"Calculate bitwise OR on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_or_not1_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_not1_saveexec_b32.json","aliases":["s_orn2_saveexec_b32"]},{"mnemonic":"s_or_not1_saveexec_b64","slug":"s_or_not1_saveexec_b64","records":1,"summary":"Calculate bitwise OR on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_or_not1_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_not1_saveexec_b64.json","aliases":["s_orn2_saveexec_b64"]},{"mnemonic":"s_or_saveexec_b32","slug":"s_or_saveexec_b32","records":1,"summary":"Calculate bitwise OR on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_or_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_saveexec_b32.json"},{"mnemonic":"s_or_saveexec_b64","slug":"s_or_saveexec_b64","records":1,"summary":"Calculate bitwise OR on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_or_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_or_saveexec_b64.json"},{"mnemonic":"s_orn1_saveexec_b32","slug":"s_orn1_saveexec_b32","records":1,"summary":"Calculate bitwise OR on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_orn1_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_orn1_saveexec_b32.json","aliases":["s_or_not0_saveexec_b32"]},{"mnemonic":"s_orn1_saveexec_b64","slug":"s_orn1_saveexec_b64","records":1,"summary":"Calculate bitwise OR on the EXEC mask and the negation of the scalar input, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_orn1_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_orn1_saveexec_b64.json","aliases":["s_or_not0_saveexec_b64"]},{"mnemonic":"s_orn2_b32","slug":"s_orn2_b32","records":1,"summary":"Calculate bitwise OR with the first input and the negation of the second input, store the result into a scalar register and set SCC if the result is…","page":"https://instructionsets.com/amdgpu/s_orn2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_orn2_b32.json","aliases":["s_or_not1_b32"]},{"mnemonic":"s_orn2_b64","slug":"s_orn2_b64","records":1,"summary":"Calculate bitwise OR with the first input and the negation of the second input, store the result into a scalar register and set SCC if the result is…","page":"https://instructionsets.com/amdgpu/s_orn2_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_orn2_b64.json","aliases":["s_or_not1_b64"]},{"mnemonic":"s_orn2_saveexec_b32","slug":"s_orn2_saveexec_b32","records":1,"summary":"Calculate bitwise OR on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_orn2_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_orn2_saveexec_b32.json","aliases":["s_or_not1_saveexec_b32"]},{"mnemonic":"s_orn2_saveexec_b64","slug":"s_orn2_saveexec_b64","records":1,"summary":"Calculate bitwise OR on the scalar input and the negation of the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the…","page":"https://instructionsets.com/amdgpu/s_orn2_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_orn2_saveexec_b64.json","aliases":["s_or_not1_saveexec_b64"]},{"mnemonic":"s_pack_hh_b32_b16","slug":"s_pack_hh_b32_b16","records":1,"summary":"Pack two 16-bit scalar values into a scalar register.","page":"https://instructionsets.com/amdgpu/s_pack_hh_b32_b16/","api":"https://instructionsets.com/api/v1/amdgpu/s_pack_hh_b32_b16.json"},{"mnemonic":"s_pack_hl_b32_b16","slug":"s_pack_hl_b32_b16","records":1,"summary":"Pack two 16-bit scalar values into a scalar register.","page":"https://instructionsets.com/amdgpu/s_pack_hl_b32_b16/","api":"https://instructionsets.com/api/v1/amdgpu/s_pack_hl_b32_b16.json"},{"mnemonic":"s_pack_lh_b32_b16","slug":"s_pack_lh_b32_b16","records":1,"summary":"Pack two 16-bit scalar values into a scalar register.","page":"https://instructionsets.com/amdgpu/s_pack_lh_b32_b16/","api":"https://instructionsets.com/api/v1/amdgpu/s_pack_lh_b32_b16.json"},{"mnemonic":"s_pack_ll_b32_b16","slug":"s_pack_ll_b32_b16","records":1,"summary":"Pack two 16-bit scalar values into a scalar register.","page":"https://instructionsets.com/amdgpu/s_pack_ll_b32_b16/","api":"https://instructionsets.com/api/v1/amdgpu/s_pack_ll_b32_b16.json"},{"mnemonic":"s_prefetch_data","slug":"s_prefetch_data","records":1,"summary":"Prefetch data into the scalar data cache, relative to a base address provided.","page":"https://instructionsets.com/amdgpu/s_prefetch_data/","api":"https://instructionsets.com/api/v1/amdgpu/s_prefetch_data.json"},{"mnemonic":"s_prefetch_data_pc_rel","slug":"s_prefetch_data_pc_rel","records":1,"summary":"Prefetch data into the scalar data cache, relative to the current PC address.","page":"https://instructionsets.com/amdgpu/s_prefetch_data_pc_rel/","api":"https://instructionsets.com/api/v1/amdgpu/s_prefetch_data_pc_rel.json"},{"mnemonic":"s_prefetch_inst","slug":"s_prefetch_inst","records":1,"summary":"Prefetch instructions into the shader instruction cache, relative to a base address provided.","page":"https://instructionsets.com/amdgpu/s_prefetch_inst/","api":"https://instructionsets.com/api/v1/amdgpu/s_prefetch_inst.json"},{"mnemonic":"s_prefetch_inst_pc_rel","slug":"s_prefetch_inst_pc_rel","records":1,"summary":"Prefetch instructions into the shader instruction cache, relative to the current PC address.","page":"https://instructionsets.com/amdgpu/s_prefetch_inst_pc_rel/","api":"https://instructionsets.com/api/v1/amdgpu/s_prefetch_inst_pc_rel.json"},{"mnemonic":"s_quadmask_b32","slug":"s_quadmask_b32","records":1,"summary":"Reduce a pixel mask from the scalar input into a quad mask, store the result in a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_quadmask_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_quadmask_b32.json"},{"mnemonic":"s_quadmask_b64","slug":"s_quadmask_b64","records":1,"summary":"Reduce a pixel mask from the scalar input into a quad mask, store the result in a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_quadmask_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_quadmask_b64.json"},{"mnemonic":"s_rfe_b64","slug":"s_rfe_b64","records":1,"summary":"Return from the exception handler.","page":"https://instructionsets.com/amdgpu/s_rfe_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_rfe_b64.json","aliases":["s_rfe_i64"]},{"mnemonic":"s_rfe_i64","slug":"s_rfe_i64","records":1,"summary":"AMDGPU SOP1 scalar instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_rfe_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_rfe_i64.json","aliases":["s_rfe_b64"]},{"mnemonic":"s_rfe_restore_b64","slug":"s_rfe_restore_b64","records":1,"summary":"Return from exception handler and continue. This instruction may only be used within a trap handler.","page":"https://instructionsets.com/amdgpu/s_rfe_restore_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_rfe_restore_b64.json"},{"mnemonic":"s_rndne_f16","slug":"s_rndne_f16","records":1,"summary":"Round the half-precision float input to the nearest even integer and store the result in floating point format into a scalar register.","page":"https://instructionsets.com/amdgpu/s_rndne_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_rndne_f16.json"},{"mnemonic":"s_rndne_f32","slug":"s_rndne_f32","records":1,"summary":"Round the single-precision float input to the nearest even integer and store the result in floating point format into a scalar register.","page":"https://instructionsets.com/amdgpu/s_rndne_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_rndne_f32.json"},{"mnemonic":"s_round_mode","slug":"s_round_mode","records":1,"summary":"Set floating point round mode using an immediate constant.","page":"https://instructionsets.com/amdgpu/s_round_mode/","api":"https://instructionsets.com/api/v1/amdgpu/s_round_mode.json"},{"mnemonic":"s_scratch_load_dword","slug":"s_scratch_load_dword","records":1,"summary":"Load 32 bits of data from the scalar scratch aperture into a scalar register.","page":"https://instructionsets.com/amdgpu/s_scratch_load_dword/","api":"https://instructionsets.com/api/v1/amdgpu/s_scratch_load_dword.json"},{"mnemonic":"s_scratch_load_dwordx2","slug":"s_scratch_load_dwordx2","records":1,"summary":"Load 64 bits of data from the scalar scratch aperture into a scalar register.","page":"https://instructionsets.com/amdgpu/s_scratch_load_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/s_scratch_load_dwordx2.json"},{"mnemonic":"s_scratch_load_dwordx4","slug":"s_scratch_load_dwordx4","records":1,"summary":"Load 128 bits of data from the scalar scratch aperture into a scalar register.","page":"https://instructionsets.com/amdgpu/s_scratch_load_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/s_scratch_load_dwordx4.json"},{"mnemonic":"s_scratch_store_dword","slug":"s_scratch_store_dword","records":1,"summary":"Store 32 bits of data from a scalar register into the scalar scratch aperture.","page":"https://instructionsets.com/amdgpu/s_scratch_store_dword/","api":"https://instructionsets.com/api/v1/amdgpu/s_scratch_store_dword.json"},{"mnemonic":"s_scratch_store_dwordx2","slug":"s_scratch_store_dwordx2","records":1,"summary":"Store 64 bits of data from a scalar register into the scalar scratch aperture.","page":"https://instructionsets.com/amdgpu/s_scratch_store_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/s_scratch_store_dwordx2.json"},{"mnemonic":"s_scratch_store_dwordx4","slug":"s_scratch_store_dwordx4","records":1,"summary":"Store 128 bits of data from a scalar register into the scalar scratch aperture.","page":"https://instructionsets.com/amdgpu/s_scratch_store_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/s_scratch_store_dwordx4.json"},{"mnemonic":"s_sendmsg","slug":"s_sendmsg","records":1,"summary":"Send a message upstream to graphics control hardware. SIMM16[9:0] contains the message type.","page":"https://instructionsets.com/amdgpu/s_sendmsg/","api":"https://instructionsets.com/api/v1/amdgpu/s_sendmsg.json"},{"mnemonic":"s_sendmsg_rtn_b32","slug":"s_sendmsg_rtn_b32","records":1,"summary":"Send a message to upstream control hardware.","page":"https://instructionsets.com/amdgpu/s_sendmsg_rtn_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_sendmsg_rtn_b32.json"},{"mnemonic":"s_sendmsg_rtn_b64","slug":"s_sendmsg_rtn_b64","records":1,"summary":"Send a message to upstream control hardware.","page":"https://instructionsets.com/amdgpu/s_sendmsg_rtn_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_sendmsg_rtn_b64.json"},{"mnemonic":"s_sendmsghalt","slug":"s_sendmsghalt","records":1,"summary":"Send a message to upstream control hardware and then HALT the wavefront; see S_SENDMSG for details.","page":"https://instructionsets.com/amdgpu/s_sendmsghalt/","api":"https://instructionsets.com/api/v1/amdgpu/s_sendmsghalt.json"},{"mnemonic":"s_set_gpr_idx_idx","slug":"s_set_gpr_idx_idx","records":1,"summary":"Set the index used in vector GPR indexing. S_SET_GPR_IDX_ON, S_SET_GPR_IDX_OFF, S_SET_GPR_IDX_MODE and S_SET_GPR_IDX_IDX are related instructions.","page":"https://instructionsets.com/amdgpu/s_set_gpr_idx_idx/","api":"https://instructionsets.com/api/v1/amdgpu/s_set_gpr_idx_idx.json"},{"mnemonic":"s_set_gpr_idx_mode","slug":"s_set_gpr_idx_mode","records":1,"summary":"Modify the mode used for vector GPR indexing.","page":"https://instructionsets.com/amdgpu/s_set_gpr_idx_mode/","api":"https://instructionsets.com/api/v1/amdgpu/s_set_gpr_idx_mode.json"},{"mnemonic":"s_set_gpr_idx_off","slug":"s_set_gpr_idx_off","records":1,"summary":"Clear GPR indexing mode.","page":"https://instructionsets.com/amdgpu/s_set_gpr_idx_off/","api":"https://instructionsets.com/api/v1/amdgpu/s_set_gpr_idx_off.json"},{"mnemonic":"s_set_gpr_idx_on","slug":"s_set_gpr_idx_on","records":1,"summary":"Enable GPR indexing mode.","page":"https://instructionsets.com/amdgpu/s_set_gpr_idx_on/","api":"https://instructionsets.com/api/v1/amdgpu/s_set_gpr_idx_on.json"},{"mnemonic":"s_set_inst_prefetch_distance","slug":"s_set_inst_prefetch_distance","records":1,"summary":"Change instruction prefetch mode. This controls how many cachelines ahead of the current PC the shader attempts to prefetch.","page":"https://instructionsets.com/amdgpu/s_set_inst_prefetch_distance/","api":"https://instructionsets.com/api/v1/amdgpu/s_set_inst_prefetch_distance.json","aliases":["s_inst_prefetch"]},{"mnemonic":"s_set_pc_i64","slug":"s_set_pc_i64","records":1,"summary":"AMDGPU SOP1 scalar instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_set_pc_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_set_pc_i64.json","aliases":["s_setpc_b64"]},{"mnemonic":"s_set_valu_coexec_mode","slug":"s_set_valu_coexec_mode","records":1,"summary":"Set the vector ALU co-execution mode to the value encoded in SIMM16[1:0] for the next VALU instruction.","page":"https://instructionsets.com/amdgpu/s_set_valu_coexec_mode/","api":"https://instructionsets.com/api/v1/amdgpu/s_set_valu_coexec_mode.json"},{"mnemonic":"s_set_vgpr_msb","slug":"s_set_vgpr_msb","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_set_vgpr_msb/","api":"https://instructionsets.com/api/v1/amdgpu/s_set_vgpr_msb.json"},{"mnemonic":"s_sethalt","slug":"s_sethalt","records":1,"summary":"Set or clear the HALT status bit.","page":"https://instructionsets.com/amdgpu/s_sethalt/","api":"https://instructionsets.com/api/v1/amdgpu/s_sethalt.json"},{"mnemonic":"s_setkill","slug":"s_setkill","records":1,"summary":"Kill this wave if the least significant bit of the immediate constant is 1. Used primarily for debugging kill wave host command behavior.","page":"https://instructionsets.com/amdgpu/s_setkill/","api":"https://instructionsets.com/api/v1/amdgpu/s_setkill.json"},{"mnemonic":"s_setpc_b64","slug":"s_setpc_b64","records":1,"summary":"Jump to an address specified in a scalar register. The argument is a byte address of the instruction to jump to.","page":"https://instructionsets.com/amdgpu/s_setpc_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_setpc_b64.json","aliases":["s_set_pc_i64"]},{"mnemonic":"s_setprio","slug":"s_setprio","records":1,"summary":"Change wave user priority.","page":"https://instructionsets.com/amdgpu/s_setprio/","api":"https://instructionsets.com/api/v1/amdgpu/s_setprio.json"},{"mnemonic":"s_setprio_inc_wg","slug":"s_setprio_inc_wg","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_setprio_inc_wg/","api":"https://instructionsets.com/api/v1/amdgpu/s_setprio_inc_wg.json"},{"mnemonic":"s_setreg_b32","slug":"s_setreg_b32","records":1,"summary":"Write some or all of the LSBs of source argument into a hardware register.","page":"https://instructionsets.com/amdgpu/s_setreg_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_setreg_b32.json"},{"mnemonic":"s_setreg_imm32_b32","slug":"s_setreg_imm32_b32","records":1,"summary":"Write some or all of the LSBs of a 32-bit literal constant into a hardware register; this instruction requires a 32-bit literal constant.","page":"https://instructionsets.com/amdgpu/s_setreg_imm32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_setreg_imm32_b32.json"},{"mnemonic":"s_setvskip","slug":"s_setvskip","records":1,"summary":"Enables or disables VSKIP mode.","page":"https://instructionsets.com/amdgpu/s_setvskip/","api":"https://instructionsets.com/api/v1/amdgpu/s_setvskip.json"},{"mnemonic":"s_sext_i32_i16","slug":"s_sext_i32_i16","records":1,"summary":"Sign extend a signed 16 bit scalar input to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_sext_i32_i16/","api":"https://instructionsets.com/api/v1/amdgpu/s_sext_i32_i16.json"},{"mnemonic":"s_sext_i32_i8","slug":"s_sext_i32_i8","records":1,"summary":"Sign extend a signed 8 bit scalar input to 32 bits and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_sext_i32_i8/","api":"https://instructionsets.com/api/v1/amdgpu/s_sext_i32_i8.json"},{"mnemonic":"s_sleep","slug":"s_sleep","records":1,"summary":"Cause a wave to sleep for up to ~8000 clocks.","page":"https://instructionsets.com/amdgpu/s_sleep/","api":"https://instructionsets.com/api/v1/amdgpu/s_sleep.json"},{"mnemonic":"s_sleep_var","slug":"s_sleep_var","records":1,"summary":"Cause a wave to sleep for up to ~8000 clocks, or to sleep until an external event wakes the wave up.","page":"https://instructionsets.com/amdgpu/s_sleep_var/","api":"https://instructionsets.com/api/v1/amdgpu/s_sleep_var.json"},{"mnemonic":"s_soft_wait_bvhcnt","slug":"s_soft_wait_bvhcnt","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_soft_wait_bvhcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_soft_wait_bvhcnt.json"},{"mnemonic":"s_soft_wait_dscnt","slug":"s_soft_wait_dscnt","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_soft_wait_dscnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_soft_wait_dscnt.json"},{"mnemonic":"s_soft_wait_kmcnt","slug":"s_soft_wait_kmcnt","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_soft_wait_kmcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_soft_wait_kmcnt.json"},{"mnemonic":"s_soft_wait_loadcnt","slug":"s_soft_wait_loadcnt","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_soft_wait_loadcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_soft_wait_loadcnt.json"},{"mnemonic":"s_soft_wait_samplecnt","slug":"s_soft_wait_samplecnt","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_soft_wait_samplecnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_soft_wait_samplecnt.json"},{"mnemonic":"s_soft_wait_storecnt","slug":"s_soft_wait_storecnt","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_soft_wait_storecnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_soft_wait_storecnt.json"},{"mnemonic":"s_soft_waitcnt","slug":"s_soft_waitcnt","records":1,"summary":"AMDGPU SOPP scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_soft_waitcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_soft_waitcnt.json"},{"mnemonic":"s_soft_waitcnt_vscnt","slug":"s_soft_waitcnt_vscnt","records":1,"summary":"AMDGPU SOPK scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_soft_waitcnt_vscnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_soft_waitcnt_vscnt.json"},{"mnemonic":"s_store_dword","slug":"s_store_dword","records":1,"summary":"Store 32 bits of data from a scalar register into the scalar memory.","page":"https://instructionsets.com/amdgpu/s_store_dword/","api":"https://instructionsets.com/api/v1/amdgpu/s_store_dword.json"},{"mnemonic":"s_store_dwordx2","slug":"s_store_dwordx2","records":1,"summary":"Store 64 bits of data from a scalar register into the scalar memory.","page":"https://instructionsets.com/amdgpu/s_store_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/s_store_dwordx2.json"},{"mnemonic":"s_store_dwordx4","slug":"s_store_dwordx4","records":1,"summary":"Store 128 bits of data from a scalar register into the scalar memory.","page":"https://instructionsets.com/amdgpu/s_store_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/s_store_dwordx4.json"},{"mnemonic":"s_sub_co_ci_u32","slug":"s_sub_co_ci_u32","records":1,"summary":"Subtract the second unsigned 32-bit integer input from the first input, subtract the carry-in bit, store the result into a scalar register and store…","page":"https://instructionsets.com/amdgpu/s_sub_co_ci_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_sub_co_ci_u32.json","aliases":["s_subb_u32"]},{"mnemonic":"s_sub_co_i32","slug":"s_sub_co_i32","records":1,"summary":"Subtract the second signed 32-bit integer input from the first input, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_sub_co_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_sub_co_i32.json","aliases":["s_sub_i32"]},{"mnemonic":"s_sub_co_u32","slug":"s_sub_co_u32","records":1,"summary":"Subtract the second unsigned 32-bit integer input from the first input, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_sub_co_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_sub_co_u32.json","aliases":["s_sub_u32"]},{"mnemonic":"s_sub_f16","slug":"s_sub_f16","records":1,"summary":"Subtract the second floating point input from the first input and store the result in a scalar register.","page":"https://instructionsets.com/amdgpu/s_sub_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_sub_f16.json"},{"mnemonic":"s_sub_f32","slug":"s_sub_f32","records":1,"summary":"Subtract the second floating point input from the first input and store the result in a scalar register.","page":"https://instructionsets.com/amdgpu/s_sub_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_sub_f32.json"},{"mnemonic":"s_sub_i32","slug":"s_sub_i32","records":1,"summary":"Subtract the second signed 32-bit integer input from the first input, store the result into a scalar register and store the carry-out bit into SCC.","page":"https://instructionsets.com/amdgpu/s_sub_i32/","api":"https://instructionsets.com/api/v1/amdgpu/s_sub_i32.json","aliases":["s_sub_co_i32"]},{"mnemonic":"s_sub_nc_u64","slug":"s_sub_nc_u64","records":1,"summary":"Subtract the second unsigned 64-bit integer input from the first input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/s_sub_nc_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_sub_nc_u64.json","aliases":["s_sub_u64"]},{"mnemonic":"s_sub_u32","slug":"s_sub_u32","records":1,"summary":"Subtract two 32-bit unsigned scalar operands, wavefront-uniform.","page":"https://instructionsets.com/amdgpu/s_sub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_sub_u32.json","aliases":["s_sub_co_u32"]},{"mnemonic":"s_sub_u64","slug":"s_sub_u64","records":1,"summary":"AMDGPU SOP2 scalar instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_sub_u64/","api":"https://instructionsets.com/api/v1/amdgpu/s_sub_u64.json","aliases":["s_sub_nc_u64"]},{"mnemonic":"s_subb_u32","slug":"s_subb_u32","records":1,"summary":"Subtract the second unsigned 32-bit integer input from the first input, subtract the carry-in bit, store the result into a scalar register and store…","page":"https://instructionsets.com/amdgpu/s_subb_u32/","api":"https://instructionsets.com/api/v1/amdgpu/s_subb_u32.json","aliases":["s_sub_co_ci_u32"]},{"mnemonic":"s_subvector_loop_begin","slug":"s_subvector_loop_begin","records":1,"summary":"Begin execution of a subvector block of code.","page":"https://instructionsets.com/amdgpu/s_subvector_loop_begin/","api":"https://instructionsets.com/api/v1/amdgpu/s_subvector_loop_begin.json"},{"mnemonic":"s_subvector_loop_end","slug":"s_subvector_loop_end","records":1,"summary":"End execution of a subvector block of code.","page":"https://instructionsets.com/amdgpu/s_subvector_loop_end/","api":"https://instructionsets.com/api/v1/amdgpu/s_subvector_loop_end.json"},{"mnemonic":"s_swap_pc_i64","slug":"s_swap_pc_i64","records":1,"summary":"AMDGPU SOP1 scalar instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_swap_pc_i64/","api":"https://instructionsets.com/api/v1/amdgpu/s_swap_pc_i64.json","aliases":["s_swappc_b64"]},{"mnemonic":"s_swappc_b64","slug":"s_swappc_b64","records":1,"summary":"Store the address of the next instruction to a scalar register and then jump to an address specified in the scalar input.","page":"https://instructionsets.com/amdgpu/s_swappc_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_swappc_b64.json","aliases":["s_swap_pc_i64"]},{"mnemonic":"s_trap","slug":"s_trap","records":1,"summary":"Enter the trap handler.","page":"https://instructionsets.com/amdgpu/s_trap/","api":"https://instructionsets.com/api/v1/amdgpu/s_trap.json"},{"mnemonic":"s_trunc_f16","slug":"s_trunc_f16","records":1,"summary":"Compute the integer part of a half-precision float input using round toward zero semantics and store the result in floating point format into a…","page":"https://instructionsets.com/amdgpu/s_trunc_f16/","api":"https://instructionsets.com/api/v1/amdgpu/s_trunc_f16.json"},{"mnemonic":"s_trunc_f32","slug":"s_trunc_f32","records":1,"summary":"Compute the integer part of a single-precision float input using round toward zero semantics and store the result in floating point format into a…","page":"https://instructionsets.com/amdgpu/s_trunc_f32/","api":"https://instructionsets.com/api/v1/amdgpu/s_trunc_f32.json"},{"mnemonic":"s_ttracedata","slug":"s_ttracedata","records":1,"summary":"Send M0 as user data to the thread trace stream.","page":"https://instructionsets.com/amdgpu/s_ttracedata/","api":"https://instructionsets.com/api/v1/amdgpu/s_ttracedata.json"},{"mnemonic":"s_ttracedata_imm","slug":"s_ttracedata_imm","records":1,"summary":"Send SIMM16[7:0] as user data to the thread trace stream.","page":"https://instructionsets.com/amdgpu/s_ttracedata_imm/","api":"https://instructionsets.com/api/v1/amdgpu/s_ttracedata_imm.json"},{"mnemonic":"s_version","slug":"s_version","records":1,"summary":"Do nothing. This opcode is used to specify the microcode version for tools that interpret shader microcode.","page":"https://instructionsets.com/amdgpu/s_version/","api":"https://instructionsets.com/api/v1/amdgpu/s_version.json"},{"mnemonic":"s_wait_alu","slug":"s_wait_alu","records":1,"summary":"Wait for one or more ALU-centric counters to fall below specified values.","page":"https://instructionsets.com/amdgpu/s_wait_alu/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_alu.json","aliases":["s_waitcnt_depctr"]},{"mnemonic":"s_wait_asynccnt","slug":"s_wait_asynccnt","records":1,"summary":"Wait until ASYNCCNT is less than or equal to SIMM16[5:0].","page":"https://instructionsets.com/amdgpu/s_wait_asynccnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_asynccnt.json"},{"mnemonic":"s_wait_bvhcnt","slug":"s_wait_bvhcnt","records":1,"summary":"Wait until BVHCNT is less than or equal to SIMM16[2:0].","page":"https://instructionsets.com/amdgpu/s_wait_bvhcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_bvhcnt.json"},{"mnemonic":"s_wait_dscnt","slug":"s_wait_dscnt","records":1,"summary":"Wait until DSCNT is less than or equal to SIMM16[5:0].","page":"https://instructionsets.com/amdgpu/s_wait_dscnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_dscnt.json"},{"mnemonic":"s_wait_event","slug":"s_wait_event","records":1,"summary":"Wait for an event to occur or a condition to be satisfied before continuing. The SIMM16 argument specifies which event(s) to wait on.","page":"https://instructionsets.com/amdgpu/s_wait_event/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_event.json"},{"mnemonic":"s_wait_expcnt","slug":"s_wait_expcnt","records":1,"summary":"Wait until EXPCNT is less than or equal to SIMM16[2:0].","page":"https://instructionsets.com/amdgpu/s_wait_expcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_expcnt.json"},{"mnemonic":"s_wait_idle","slug":"s_wait_idle","records":1,"summary":"Wait for all activity in the wave to be complete (all dependency and memory counters at zero).","page":"https://instructionsets.com/amdgpu/s_wait_idle/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_idle.json"},{"mnemonic":"s_wait_kmcnt","slug":"s_wait_kmcnt","records":1,"summary":"Wait until KMCNT is less than or equal to SIMM16[4:0].","page":"https://instructionsets.com/amdgpu/s_wait_kmcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_kmcnt.json"},{"mnemonic":"s_wait_loadcnt","slug":"s_wait_loadcnt","records":1,"summary":"Wait until LOADCNT is less than or equal to SIMM16[5:0].","page":"https://instructionsets.com/amdgpu/s_wait_loadcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_loadcnt.json"},{"mnemonic":"s_wait_loadcnt_dscnt","slug":"s_wait_loadcnt_dscnt","records":1,"summary":"Wait until LOADCNT is less than or equal to SIMM16[13:8] and DSCNT is less than or equal to SIMM16[5:0].","page":"https://instructionsets.com/amdgpu/s_wait_loadcnt_dscnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_loadcnt_dscnt.json"},{"mnemonic":"s_wait_samplecnt","slug":"s_wait_samplecnt","records":1,"summary":"Wait until SAMPLECNT is less than or equal to SIMM16[5:0].","page":"https://instructionsets.com/amdgpu/s_wait_samplecnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_samplecnt.json"},{"mnemonic":"s_wait_storecnt","slug":"s_wait_storecnt","records":1,"summary":"Wait until STORECNT is less than or equal to SIMM16[5:0].","page":"https://instructionsets.com/amdgpu/s_wait_storecnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_storecnt.json"},{"mnemonic":"s_wait_storecnt_dscnt","slug":"s_wait_storecnt_dscnt","records":1,"summary":"Wait until STORECNT is less than or equal to SIMM16[13:8] and DSCNT is less than or equal to SIMM16[5:0].","page":"https://instructionsets.com/amdgpu/s_wait_storecnt_dscnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_storecnt_dscnt.json"},{"mnemonic":"s_wait_tensorcnt","slug":"s_wait_tensorcnt","records":1,"summary":"Wait until TENSORCNT is less than or equal to SIMM16[5:0].","page":"https://instructionsets.com/amdgpu/s_wait_tensorcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_tensorcnt.json"},{"mnemonic":"s_wait_xcnt","slug":"s_wait_xcnt","records":1,"summary":"Wait until XCNT is less than or equal to SIMM16[5:0].","page":"https://instructionsets.com/amdgpu/s_wait_xcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_wait_xcnt.json"},{"mnemonic":"s_waitcnt","slug":"s_waitcnt","records":1,"summary":"Wait for the counts of outstanding local data share, vector memory and export instructions to be at or below the specified levels.","page":"https://instructionsets.com/amdgpu/s_waitcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_waitcnt.json"},{"mnemonic":"s_waitcnt_depctr","slug":"s_waitcnt_depctr","records":1,"summary":"Wait for one or more ALU-centric counters to fall below specified values. Used in expert scheduling mode.","page":"https://instructionsets.com/amdgpu/s_waitcnt_depctr/","api":"https://instructionsets.com/api/v1/amdgpu/s_waitcnt_depctr.json","aliases":["s_wait_alu"]},{"mnemonic":"s_waitcnt_expcnt","slug":"s_waitcnt_expcnt","records":1,"summary":"Wait for the EXPCNT counter to be at or below the specified level. The EXPCNT counter tracks the number of outstanding export events.","page":"https://instructionsets.com/amdgpu/s_waitcnt_expcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_waitcnt_expcnt.json"},{"mnemonic":"s_waitcnt_lgkmcnt","slug":"s_waitcnt_lgkmcnt","records":1,"summary":"Wait for the LGKMCNT counter to be at or below the specified level.","page":"https://instructionsets.com/amdgpu/s_waitcnt_lgkmcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_waitcnt_lgkmcnt.json"},{"mnemonic":"s_waitcnt_vmcnt","slug":"s_waitcnt_vmcnt","records":1,"summary":"Wait for the VMCNT counter to be at or below the specified level.","page":"https://instructionsets.com/amdgpu/s_waitcnt_vmcnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_waitcnt_vmcnt.json"},{"mnemonic":"s_waitcnt_vscnt","slug":"s_waitcnt_vscnt","records":1,"summary":"Wait for the VSCNT counter to be at or below the specified level.","page":"https://instructionsets.com/amdgpu/s_waitcnt_vscnt/","api":"https://instructionsets.com/api/v1/amdgpu/s_waitcnt_vscnt.json"},{"mnemonic":"s_wakeup","slug":"s_wakeup","records":1,"summary":"Allow a wave to 'ping' all the other waves in its threadgroup to force them to wake up early from an S_SLEEP instruction.","page":"https://instructionsets.com/amdgpu/s_wakeup/","api":"https://instructionsets.com/api/v1/amdgpu/s_wakeup.json"},{"mnemonic":"s_wakeup_barrier","slug":"s_wakeup_barrier","records":1,"summary":"AMDGPU SOP1 scalar instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/s_wakeup_barrier/","api":"https://instructionsets.com/api/v1/amdgpu/s_wakeup_barrier.json"},{"mnemonic":"s_wqm_b32","slug":"s_wqm_b32","records":1,"summary":"Given an active pixel mask in a scalar input, calculate whole quad mode mask for that input, store the result into a scalar register and set SCC iff…","page":"https://instructionsets.com/amdgpu/s_wqm_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_wqm_b32.json"},{"mnemonic":"s_wqm_b64","slug":"s_wqm_b64","records":1,"summary":"Given an active pixel mask in a scalar input, calculate whole quad mode mask for that input, store the result into a scalar register and set SCC iff…","page":"https://instructionsets.com/amdgpu/s_wqm_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_wqm_b64.json"},{"mnemonic":"s_xnor_b32","slug":"s_xnor_b32","records":1,"summary":"Calculate bitwise XNOR on two scalar inputs, store the result into a scalar register and set SCC if the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_xnor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_xnor_b32.json"},{"mnemonic":"s_xnor_b64","slug":"s_xnor_b64","records":1,"summary":"Calculate bitwise XNOR on two scalar inputs, store the result into a scalar register and set SCC if the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_xnor_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_xnor_b64.json"},{"mnemonic":"s_xnor_saveexec_b32","slug":"s_xnor_saveexec_b32","records":1,"summary":"Calculate bitwise XNOR on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_xnor_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_xnor_saveexec_b32.json"},{"mnemonic":"s_xnor_saveexec_b64","slug":"s_xnor_saveexec_b64","records":1,"summary":"Calculate bitwise XNOR on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_xnor_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_xnor_saveexec_b64.json"},{"mnemonic":"s_xor_b32","slug":"s_xor_b32","records":1,"summary":"Calculate bitwise XOR on two scalar inputs, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_xor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_xor_b32.json"},{"mnemonic":"s_xor_b64","slug":"s_xor_b64","records":1,"summary":"Calculate bitwise XOR on two scalar inputs, store the result into a scalar register and set SCC iff the result is nonzero.","page":"https://instructionsets.com/amdgpu/s_xor_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_xor_b64.json"},{"mnemonic":"s_xor_saveexec_b32","slug":"s_xor_saveexec_b32","records":1,"summary":"Calculate bitwise XOR on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_xor_saveexec_b32/","api":"https://instructionsets.com/api/v1/amdgpu/s_xor_saveexec_b32.json"},{"mnemonic":"s_xor_saveexec_b64","slug":"s_xor_saveexec_b64","records":1,"summary":"Calculate bitwise XOR on the scalar input and the EXEC mask, store the calculated result into the EXEC mask, set SCC iff the calculated result is…","page":"https://instructionsets.com/amdgpu/s_xor_saveexec_b64/","api":"https://instructionsets.com/api/v1/amdgpu/s_xor_saveexec_b64.json"},{"mnemonic":"scratch_load_block","slug":"scratch_load_block","records":1,"summary":"Load a block of data from the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_load_block/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_block.json"},{"mnemonic":"scratch_load_dword","slug":"scratch_load_dword","records":1,"summary":"Load 32 bits of data from the scratch aperture into a vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_dword/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_dword.json","aliases":["scratch_load_b32"]},{"mnemonic":"scratch_load_dwordx2","slug":"scratch_load_dwordx2","records":1,"summary":"Load 64 bits of data from the scratch aperture into a vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_dwordx2.json","aliases":["scratch_load_b64"]},{"mnemonic":"scratch_load_dwordx3","slug":"scratch_load_dwordx3","records":1,"summary":"Load 96 bits of data from the scratch aperture into a vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_dwordx3/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_dwordx3.json","aliases":["scratch_load_b96"]},{"mnemonic":"scratch_load_dwordx4","slug":"scratch_load_dwordx4","records":1,"summary":"Load 128 bits of data from the scratch aperture into a vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_dwordx4.json","aliases":["scratch_load_b128"]},{"mnemonic":"scratch_load_lds_dword","slug":"scratch_load_lds_dword","records":1,"summary":"Load 32 bits of untyped data from the scratch aperture and store the result into a data share.","page":"https://instructionsets.com/amdgpu/scratch_load_lds_dword/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_lds_dword.json"},{"mnemonic":"scratch_load_lds_sbyte","slug":"scratch_load_lds_sbyte","records":1,"summary":"Load 8 bits of untyped data from the scratch aperture, sign extend to 32 bits and store the result into a data share.","page":"https://instructionsets.com/amdgpu/scratch_load_lds_sbyte/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_lds_sbyte.json"},{"mnemonic":"scratch_load_lds_sshort","slug":"scratch_load_lds_sshort","records":1,"summary":"Load 16 bits of untyped data from the scratch aperture, sign extend to 32 bits and store the result into a data share.","page":"https://instructionsets.com/amdgpu/scratch_load_lds_sshort/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_lds_sshort.json"},{"mnemonic":"scratch_load_lds_ubyte","slug":"scratch_load_lds_ubyte","records":1,"summary":"Load 8 bits of untyped data from the scratch aperture, zero extend to 32 bits and store the result into a data share.","page":"https://instructionsets.com/amdgpu/scratch_load_lds_ubyte/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_lds_ubyte.json"},{"mnemonic":"scratch_load_lds_ushort","slug":"scratch_load_lds_ushort","records":1,"summary":"Load 16 bits of untyped data from the scratch aperture, zero extend to 32 bits and store the result into a data share.","page":"https://instructionsets.com/amdgpu/scratch_load_lds_ushort/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_lds_ushort.json"},{"mnemonic":"scratch_load_sbyte","slug":"scratch_load_sbyte","records":1,"summary":"Load 8 bits of signed data from the scratch aperture, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_sbyte/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_sbyte.json","aliases":["scratch_load_i8"]},{"mnemonic":"scratch_load_sbyte_d16","slug":"scratch_load_sbyte_d16","records":1,"summary":"Load 8 bits of signed data from the scratch aperture, sign extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_sbyte_d16/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_sbyte_d16.json","aliases":["scratch_load_d16_i8"]},{"mnemonic":"scratch_load_sbyte_d16_hi","slug":"scratch_load_sbyte_d16_hi","records":1,"summary":"Load 8 bits of signed data from the scratch aperture, sign extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_sbyte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_sbyte_d16_hi.json","aliases":["scratch_load_d16_hi_i8"]},{"mnemonic":"scratch_load_short_d16","slug":"scratch_load_short_d16","records":1,"summary":"Load 16 bits of unsigned data from the scratch aperture and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_short_d16/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_short_d16.json","aliases":["scratch_load_d16_b16"]},{"mnemonic":"scratch_load_short_d16_hi","slug":"scratch_load_short_d16_hi","records":1,"summary":"Load 16 bits of unsigned data from the scratch aperture and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_short_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_short_d16_hi.json","aliases":["scratch_load_d16_hi_b16"]},{"mnemonic":"scratch_load_sshort","slug":"scratch_load_sshort","records":1,"summary":"Load 16 bits of signed data from the scratch aperture, sign extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_sshort/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_sshort.json","aliases":["scratch_load_i16"]},{"mnemonic":"scratch_load_ubyte","slug":"scratch_load_ubyte","records":1,"summary":"Load 8 bits of unsigned data from the scratch aperture, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_ubyte/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_ubyte.json","aliases":["scratch_load_u8"]},{"mnemonic":"scratch_load_ubyte_d16","slug":"scratch_load_ubyte_d16","records":1,"summary":"Load 8 bits of unsigned data from the scratch aperture, zero extend to 16 bits and store the result into the low 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_ubyte_d16/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_ubyte_d16.json","aliases":["scratch_load_d16_u8"]},{"mnemonic":"scratch_load_ubyte_d16_hi","slug":"scratch_load_ubyte_d16_hi","records":1,"summary":"Load 8 bits of unsigned data from the scratch aperture, zero extend to 16 bits and store the result into the high 16 bits of a 32-bit vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_ubyte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_ubyte_d16_hi.json","aliases":["scratch_load_d16_hi_u8"]},{"mnemonic":"scratch_load_ushort","slug":"scratch_load_ushort","records":1,"summary":"Load 16 bits of unsigned data from the scratch aperture, zero extend to 32 bits and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/scratch_load_ushort/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_load_ushort.json","aliases":["scratch_load_u16"]},{"mnemonic":"scratch_store_block","slug":"scratch_store_block","records":1,"summary":"Store a block of data to the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_store_block/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_store_block.json"},{"mnemonic":"scratch_store_byte","slug":"scratch_store_byte","records":1,"summary":"Store 8 bits of data from a vector register into the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_store_byte/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_store_byte.json","aliases":["scratch_store_b8"]},{"mnemonic":"scratch_store_byte_d16_hi","slug":"scratch_store_byte_d16_hi","records":1,"summary":"Store 8 bits of data from the high 16 bits of a 32-bit vector register into the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_store_byte_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_store_byte_d16_hi.json","aliases":["scratch_store_d16_hi_b8"]},{"mnemonic":"scratch_store_dword","slug":"scratch_store_dword","records":1,"summary":"Store 32 bits of data from vector input registers into the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_store_dword/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_store_dword.json","aliases":["scratch_store_b32"]},{"mnemonic":"scratch_store_dwordx2","slug":"scratch_store_dwordx2","records":1,"summary":"Store 64 bits of data from vector input registers into the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_store_dwordx2/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_store_dwordx2.json","aliases":["scratch_store_b64"]},{"mnemonic":"scratch_store_dwordx3","slug":"scratch_store_dwordx3","records":1,"summary":"Store 96 bits of data from vector input registers into the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_store_dwordx3/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_store_dwordx3.json","aliases":["scratch_store_b96"]},{"mnemonic":"scratch_store_dwordx4","slug":"scratch_store_dwordx4","records":1,"summary":"Store 128 bits of data from vector input registers into the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_store_dwordx4/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_store_dwordx4.json","aliases":["scratch_store_b128"]},{"mnemonic":"scratch_store_short","slug":"scratch_store_short","records":1,"summary":"Store 16 bits of data from a vector register into the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_store_short/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_store_short.json","aliases":["scratch_store_b16"]},{"mnemonic":"scratch_store_short_d16_hi","slug":"scratch_store_short_d16_hi","records":1,"summary":"Store 16 bits of data from the high 16 bits of a 32-bit vector register into the scratch aperture.","page":"https://instructionsets.com/amdgpu/scratch_store_short_d16_hi/","api":"https://instructionsets.com/api/v1/amdgpu/scratch_store_short_d16_hi.json","aliases":["scratch_store_d16_hi_b16"]},{"mnemonic":"tbuffer_load_d16_format_x","slug":"tbuffer_load_d16_format_x","records":1,"summary":"Load 1-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/tbuffer_load_d16_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_d16_format_x.json","aliases":["tbuffer_load_format_d16_x"]},{"mnemonic":"tbuffer_load_d16_format_xy","slug":"tbuffer_load_d16_format_xy","records":1,"summary":"Load 2-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/tbuffer_load_d16_format_xy/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_d16_format_xy.json","aliases":["tbuffer_load_format_d16_xy"]},{"mnemonic":"tbuffer_load_d16_format_xyz","slug":"tbuffer_load_d16_format_xyz","records":1,"summary":"Load 3-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/tbuffer_load_d16_format_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_d16_format_xyz.json","aliases":["tbuffer_load_format_d16_xyz"]},{"mnemonic":"tbuffer_load_d16_format_xyzw","slug":"tbuffer_load_d16_format_xyzw","records":1,"summary":"Load 4-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/tbuffer_load_d16_format_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_d16_format_xyzw.json","aliases":["tbuffer_load_format_d16_xyzw"]},{"mnemonic":"tbuffer_load_format_d16_x","slug":"tbuffer_load_format_d16_x","records":1,"summary":"Load 1-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/tbuffer_load_format_d16_x/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_format_d16_x.json","aliases":["tbuffer_load_d16_format_x"]},{"mnemonic":"tbuffer_load_format_d16_xy","slug":"tbuffer_load_format_d16_xy","records":1,"summary":"Load 2-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/tbuffer_load_format_d16_xy/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_format_d16_xy.json","aliases":["tbuffer_load_d16_format_xy"]},{"mnemonic":"tbuffer_load_format_d16_xyz","slug":"tbuffer_load_format_d16_xyz","records":1,"summary":"Load 3-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/tbuffer_load_format_d16_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_format_d16_xyz.json","aliases":["tbuffer_load_d16_format_xyz"]},{"mnemonic":"tbuffer_load_format_d16_xyzw","slug":"tbuffer_load_format_d16_xyzw","records":1,"summary":"Load 4-component formatted data from a buffer surface, convert the data to packed 16 bit integral or floating point format, then store the result…","page":"https://instructionsets.com/amdgpu/tbuffer_load_format_d16_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_format_d16_xyzw.json","aliases":["tbuffer_load_d16_format_xyzw"]},{"mnemonic":"tbuffer_load_format_x","slug":"tbuffer_load_format_x","records":1,"summary":"Load 1-component formatted data from a buffer surface, convert the data to 32 bit integral or floating point format, then store the result into a…","page":"https://instructionsets.com/amdgpu/tbuffer_load_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_format_x.json"},{"mnemonic":"tbuffer_load_format_xy","slug":"tbuffer_load_format_xy","records":1,"summary":"Load 2-component formatted data from a buffer surface, convert the data to 32 bit integral or floating point format, then store the result into a…","page":"https://instructionsets.com/amdgpu/tbuffer_load_format_xy/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_format_xy.json"},{"mnemonic":"tbuffer_load_format_xyz","slug":"tbuffer_load_format_xyz","records":1,"summary":"Load 3-component formatted data from a buffer surface, convert the data to 32 bit integral or floating point format, then store the result into a…","page":"https://instructionsets.com/amdgpu/tbuffer_load_format_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_format_xyz.json"},{"mnemonic":"tbuffer_load_format_xyzw","slug":"tbuffer_load_format_xyzw","records":1,"summary":"Load 4-component formatted data from a buffer surface, convert the data to 32 bit integral or floating point format, then store the result into a…","page":"https://instructionsets.com/amdgpu/tbuffer_load_format_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_load_format_xyzw.json"},{"mnemonic":"tbuffer_store_d16_format_x","slug":"tbuffer_store_d16_format_x","records":1,"summary":"Convert 16 bits of data from vector input registers into 1-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_d16_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_d16_format_x.json","aliases":["tbuffer_store_format_d16_x"]},{"mnemonic":"tbuffer_store_d16_format_xy","slug":"tbuffer_store_d16_format_xy","records":1,"summary":"Convert 32 bits of data from vector input registers into 2-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_d16_format_xy/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_d16_format_xy.json","aliases":["tbuffer_store_format_d16_xy"]},{"mnemonic":"tbuffer_store_d16_format_xyz","slug":"tbuffer_store_d16_format_xyz","records":1,"summary":"Convert 48 bits of data from vector input registers into 3-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_d16_format_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_d16_format_xyz.json","aliases":["tbuffer_store_format_d16_xyz"]},{"mnemonic":"tbuffer_store_d16_format_xyzw","slug":"tbuffer_store_d16_format_xyzw","records":1,"summary":"Convert 64 bits of data from vector input registers into 4-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_d16_format_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_d16_format_xyzw.json","aliases":["tbuffer_store_format_d16_xyzw"]},{"mnemonic":"tbuffer_store_format_d16_x","slug":"tbuffer_store_format_d16_x","records":1,"summary":"Convert 16 bits of data from vector input registers into 1-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_format_d16_x/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_format_d16_x.json","aliases":["tbuffer_store_d16_format_x"]},{"mnemonic":"tbuffer_store_format_d16_xy","slug":"tbuffer_store_format_d16_xy","records":1,"summary":"Convert 32 bits of data from vector input registers into 2-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_format_d16_xy/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_format_d16_xy.json","aliases":["tbuffer_store_d16_format_xy"]},{"mnemonic":"tbuffer_store_format_d16_xyz","slug":"tbuffer_store_format_d16_xyz","records":1,"summary":"Convert 48 bits of data from vector input registers into 3-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_format_d16_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_format_d16_xyz.json","aliases":["tbuffer_store_d16_format_xyz"]},{"mnemonic":"tbuffer_store_format_d16_xyzw","slug":"tbuffer_store_format_d16_xyzw","records":1,"summary":"Convert 64 bits of data from vector input registers into 4-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_format_d16_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_format_d16_xyzw.json","aliases":["tbuffer_store_d16_format_xyzw"]},{"mnemonic":"tbuffer_store_format_x","slug":"tbuffer_store_format_x","records":1,"summary":"Convert 32 bits of data from vector input registers into 1-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_format_x/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_format_x.json"},{"mnemonic":"tbuffer_store_format_xy","slug":"tbuffer_store_format_xy","records":1,"summary":"Convert 64 bits of data from vector input registers into 2-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_format_xy/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_format_xy.json"},{"mnemonic":"tbuffer_store_format_xyz","slug":"tbuffer_store_format_xyz","records":1,"summary":"Convert 96 bits of data from vector input registers into 3-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_format_xyz/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_format_xyz.json"},{"mnemonic":"tbuffer_store_format_xyzw","slug":"tbuffer_store_format_xyzw","records":1,"summary":"Convert 128 bits of data from vector input registers into 4-component formatted data and store the data into a buffer surface.","page":"https://instructionsets.com/amdgpu/tbuffer_store_format_xyzw/","api":"https://instructionsets.com/api/v1/amdgpu/tbuffer_store_format_xyzw.json"},{"mnemonic":"tensor_load_to_lds","slug":"tensor_load_to_lds","records":1,"summary":"DMA instruction to load tensor data from Global to LDS.","page":"https://instructionsets.com/amdgpu/tensor_load_to_lds/","api":"https://instructionsets.com/api/v1/amdgpu/tensor_load_to_lds.json"},{"mnemonic":"tensor_save","slug":"tensor_save","records":1,"summary":"AMDGPU FLAT vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/tensor_save/","api":"https://instructionsets.com/api/v1/amdgpu/tensor_save.json"},{"mnemonic":"tensor_stop","slug":"tensor_stop","records":1,"summary":"AMDGPU FLAT vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/tensor_stop/","api":"https://instructionsets.com/api/v1/amdgpu/tensor_stop.json"},{"mnemonic":"tensor_store_from_lds","slug":"tensor_store_from_lds","records":1,"summary":"DMA instruction to store tensor data to Global from LDS.","page":"https://instructionsets.com/amdgpu/tensor_store_from_lds/","api":"https://instructionsets.com/api/v1/amdgpu/tensor_store_from_lds.json"},{"mnemonic":"v_accvgpr_mov_b32","slug":"v_accvgpr_mov_b32","records":1,"summary":"Move data from one accumulator register to another accumulator register.","page":"https://instructionsets.com/amdgpu/v_accvgpr_mov_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_accvgpr_mov_b32.json"},{"mnemonic":"v_accvgpr_read_b32","slug":"v_accvgpr_read_b32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_accvgpr_read_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_accvgpr_read_b32.json","aliases":["v_accvgpr_read"]},{"mnemonic":"v_accvgpr_write_b32","slug":"v_accvgpr_write_b32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_accvgpr_write_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_accvgpr_write_b32.json","aliases":["v_accvgpr_write"]},{"mnemonic":"v_add3_u32","slug":"v_add3_u32","records":1,"summary":"Add three unsigned inputs and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_add3_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add3_u32.json","aliases":["v_add3_nc_u32"]},{"mnemonic":"v_add_co_ci_u32","slug":"v_add_co_ci_u32","records":1,"summary":"Add two unsigned 32-bit integer inputs and a bit from a carry-in mask, store the result into a vector register and store the carry-out mask into a…","page":"https://instructionsets.com/amdgpu/v_add_co_ci_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_co_ci_u32.json"},{"mnemonic":"v_add_co_u32","slug":"v_add_co_u32","records":1,"summary":"Add two unsigned 32-bit integer inputs, store the result into a vector register and store the carry-out mask into a scalar register.","page":"https://instructionsets.com/amdgpu/v_add_co_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_co_u32.json"},{"mnemonic":"v_add_f16","slug":"v_add_f16","records":1,"summary":"Add two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_add_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_f16.json"},{"mnemonic":"v_add_f32","slug":"v_add_f32","records":1,"summary":"Per-lane single-precision floating-point add.","page":"https://instructionsets.com/amdgpu/v_add_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_f32.json"},{"mnemonic":"v_add_f64","slug":"v_add_f64","records":1,"summary":"Add two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_add_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_f64.json"},{"mnemonic":"v_add_f64_pseudo","slug":"v_add_f64_pseudo","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_add_f64_pseudo/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_f64_pseudo.json"},{"mnemonic":"v_add_i16","slug":"v_add_i16","records":1,"summary":"Add two signed 16-bit integer inputs and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_add_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_i16.json"},{"mnemonic":"v_add_i32","slug":"v_add_i32","records":1,"summary":"Add two signed 32-bit integer inputs and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_add_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_i32.json"},{"mnemonic":"v_add_lshl_u32","slug":"v_add_lshl_u32","records":1,"summary":"Add the first two integer inputs, then given a shift count in the third input, calculate the logical shift left of the intermediate result, then…","page":"https://instructionsets.com/amdgpu/v_add_lshl_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_lshl_u32.json"},{"mnemonic":"v_add_max_i32","slug":"v_add_max_i32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_add_max_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_max_i32.json"},{"mnemonic":"v_add_max_u32","slug":"v_add_max_u32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_add_max_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_max_u32.json"},{"mnemonic":"v_add_min_i32","slug":"v_add_min_i32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_add_min_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_min_i32.json"},{"mnemonic":"v_add_min_u32","slug":"v_add_min_u32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_add_min_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_min_u32.json"},{"mnemonic":"v_add_nc_i16","slug":"v_add_nc_i16","records":1,"summary":"Add two signed 16-bit integer inputs and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_add_nc_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_nc_i16.json"},{"mnemonic":"v_add_nc_i32","slug":"v_add_nc_i32","records":1,"summary":"Add two signed 32-bit integer inputs and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_add_nc_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_nc_i32.json"},{"mnemonic":"v_add_nc_u16","slug":"v_add_nc_u16","records":1,"summary":"Add two unsigned 16-bit integer inputs and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_add_nc_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_nc_u16.json"},{"mnemonic":"v_add_nc_u32","slug":"v_add_nc_u32","records":1,"summary":"Add two unsigned 32-bit integer inputs and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_add_nc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_nc_u32.json"},{"mnemonic":"v_add_nc_u64","slug":"v_add_nc_u64","records":1,"summary":"AMDGPU VOP2 vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_add_nc_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_nc_u64.json"},{"mnemonic":"v_add_u16","slug":"v_add_u16","records":1,"summary":"Add two unsigned 16-bit integer inputs and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_add_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_u16.json"},{"mnemonic":"v_add_u32","slug":"v_add_u32","records":1,"summary":"Per-lane add of two 32-bit unsigned vector operands.","page":"https://instructionsets.com/amdgpu/v_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_add_u32.json"},{"mnemonic":"v_addc_co_u32","slug":"v_addc_co_u32","records":1,"summary":"Add two unsigned 32-bit integer inputs and a bit from a carry-in mask, store the result into a vector register and store the carry-out mask into a…","page":"https://instructionsets.com/amdgpu/v_addc_co_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_addc_co_u32.json"},{"mnemonic":"v_addc_u32","slug":"v_addc_u32","records":1,"summary":"AMDGPU VOP2 vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_addc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_addc_u32.json"},{"mnemonic":"v_alignbit_b32","slug":"v_alignbit_b32","records":1,"summary":"Align a 64-bit value encoded in the first two inputs to a bit position specified in the third input, then store the result into a 32-bit vector…","page":"https://instructionsets.com/amdgpu/v_alignbit_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_alignbit_b32.json"},{"mnemonic":"v_alignbit_b32_opsel","slug":"v_alignbit_b32_opsel","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_alignbit_b32_opsel/","api":"https://instructionsets.com/api/v1/amdgpu/v_alignbit_b32_opsel.json"},{"mnemonic":"v_alignbyte_b32","slug":"v_alignbyte_b32","records":1,"summary":"Align a 64-bit value encoded in the first two inputs to a byte position specified in the third input, then store the result into a 32-bit vector…","page":"https://instructionsets.com/amdgpu/v_alignbyte_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_alignbyte_b32.json"},{"mnemonic":"v_alignbyte_b32_fake16","slug":"v_alignbyte_b32_fake16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_alignbyte_b32_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_alignbyte_b32_fake16.json"},{"mnemonic":"v_alignbyte_b32_opsel","slug":"v_alignbyte_b32_opsel","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_alignbyte_b32_opsel/","api":"https://instructionsets.com/api/v1/amdgpu/v_alignbyte_b32_opsel.json"},{"mnemonic":"v_alignbyte_b32_t16","slug":"v_alignbyte_b32_t16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_alignbyte_b32_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_alignbyte_b32_t16.json"},{"mnemonic":"v_and_b16","slug":"v_and_b16","records":1,"summary":"Calculate bitwise AND on two vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_and_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_and_b16.json"},{"mnemonic":"v_and_b16_fake16","slug":"v_and_b16_fake16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_and_b16_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_and_b16_fake16.json"},{"mnemonic":"v_and_b16_t16","slug":"v_and_b16_t16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_and_b16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_and_b16_t16.json"},{"mnemonic":"v_and_b32","slug":"v_and_b32","records":1,"summary":"Calculate bitwise AND on two vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_and_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_and_b32.json"},{"mnemonic":"v_and_or_b32","slug":"v_and_or_b32","records":1,"summary":"Calculate bitwise AND on the first two vector inputs, then compute the bitwise OR of the intermediate result and the third vector input, then store…","page":"https://instructionsets.com/amdgpu/v_and_or_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_and_or_b32.json"},{"mnemonic":"v_ashr_i32","slug":"v_ashr_i32","records":1,"summary":"AMDGPU VOP2 vector instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_ashr_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_ashr_i32.json"},{"mnemonic":"v_ashr_i64","slug":"v_ashr_i64","records":1,"summary":"AMDGPU VOP3 vector instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_ashr_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_ashr_i64.json"},{"mnemonic":"v_ashr_pk_i8_i32","slug":"v_ashr_pk_i8_i32","records":1,"summary":"Given two signed 32-bit integers and a shift count, calculate the arithmetic shift right (preserving sign bit) of the two integers, saturate the two…","page":"https://instructionsets.com/amdgpu/v_ashr_pk_i8_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_ashr_pk_i8_i32.json"},{"mnemonic":"v_ashr_pk_u8_i32","slug":"v_ashr_pk_u8_i32","records":1,"summary":"Given two signed 32-bit integers and a shift count, calculate the arithmetic shift right (preserving sign bit) of the two integers, saturate the two…","page":"https://instructionsets.com/amdgpu/v_ashr_pk_u8_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_ashr_pk_u8_i32.json"},{"mnemonic":"v_ashrrev_i16","slug":"v_ashrrev_i16","records":1,"summary":"Given a shift count in the first vector input, calculate the arithmetic shift right (preserving sign bit) of the second vector input and store the…","page":"https://instructionsets.com/amdgpu/v_ashrrev_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_ashrrev_i16.json"},{"mnemonic":"v_ashrrev_i32","slug":"v_ashrrev_i32","records":1,"summary":"Given a shift count in the first vector input, calculate the arithmetic shift right (preserving sign bit) of the second vector input and store the…","page":"https://instructionsets.com/amdgpu/v_ashrrev_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_ashrrev_i32.json"},{"mnemonic":"v_ashrrev_i64","slug":"v_ashrrev_i64","records":1,"summary":"Given a shift count in the first vector input, calculate the arithmetic shift right (preserving sign bit) of the second vector input and store the…","page":"https://instructionsets.com/amdgpu/v_ashrrev_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_ashrrev_i64.json"},{"mnemonic":"v_bcnt_u32_b32","slug":"v_bcnt_u32_b32","records":1,"summary":"Per-lane accumulating population count.","page":"https://instructionsets.com/amdgpu/v_bcnt_u32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_bcnt_u32_b32.json"},{"mnemonic":"v_bfe_i32","slug":"v_bfe_i32","records":1,"summary":"Extract a signed bitfield from the first input using field offset from the second input and size from the third input, then store the result into a…","page":"https://instructionsets.com/amdgpu/v_bfe_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_bfe_i32.json"},{"mnemonic":"v_bfe_u32","slug":"v_bfe_u32","records":1,"summary":"Extract an unsigned bitfield from the first input using field offset from the second input and size from the third input, then store the result into…","page":"https://instructionsets.com/amdgpu/v_bfe_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_bfe_u32.json"},{"mnemonic":"v_bfi_b32","slug":"v_bfi_b32","records":1,"summary":"Overwrite a bitfield in the third input with a bitfield from the second input using a mask from the first input, then store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_bfi_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_bfi_b32.json"},{"mnemonic":"v_bfm_b32","slug":"v_bfm_b32","records":1,"summary":"Calculate a bitfield mask given a field offset and size and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_bfm_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_bfm_b32.json"},{"mnemonic":"v_bfrev_b32","slug":"v_bfrev_b32","records":1,"summary":"Reverse the order of bits in a vector input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_bfrev_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_bfrev_b32.json"},{"mnemonic":"v_bitop3_b16","slug":"v_bitop3_b16","records":1,"summary":"Calculate the generic bitwise operation of three 16-bit vector inputs using a truth table encoded in the instruction and store the result into a…","page":"https://instructionsets.com/amdgpu/v_bitop3_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_bitop3_b16.json"},{"mnemonic":"v_bitop3_b16_gfx1250","slug":"v_bitop3_b16_gfx1250","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_bitop3_b16_gfx1250/","api":"https://instructionsets.com/api/v1/amdgpu/v_bitop3_b16_gfx1250.json"},{"mnemonic":"v_bitop3_b32","slug":"v_bitop3_b32","records":1,"summary":"Calculate the generic bitwise operation of three 32-bit vector inputs using a truth table encoded in the instruction and store the result into a…","page":"https://instructionsets.com/amdgpu/v_bitop3_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_bitop3_b32.json"},{"mnemonic":"v_ceil_f16","slug":"v_ceil_f16","records":1,"summary":"Round the half-precision float input up to next integer and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_ceil_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_ceil_f16.json"},{"mnemonic":"v_ceil_f32","slug":"v_ceil_f32","records":1,"summary":"Round the single-precision float input up to next integer and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_ceil_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_ceil_f32.json"},{"mnemonic":"v_ceil_f64","slug":"v_ceil_f64","records":1,"summary":"Round the double-precision float input up to next integer and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_ceil_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_ceil_f64.json"},{"mnemonic":"v_clrexcp","slug":"v_clrexcp","records":1,"summary":"Clear this wave's exception state in the vector ALU.","page":"https://instructionsets.com/amdgpu/v_clrexcp/","api":"https://instructionsets.com/api/v1/amdgpu/v_clrexcp.json"},{"mnemonic":"v_cmp_class_f16","slug":"v_cmp_class_f16","records":1,"summary":"Evaluate the IEEE numeric class function specified as a 10 bit mask in the second input on the first input, a half-precision float, and set the…","page":"https://instructionsets.com/amdgpu/v_cmp_class_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_class_f16.json"},{"mnemonic":"v_cmp_class_f32","slug":"v_cmp_class_f32","records":1,"summary":"Evaluate the IEEE numeric class function specified as a 10 bit mask in the second input on the first input, a single-precision float, and set the…","page":"https://instructionsets.com/amdgpu/v_cmp_class_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_class_f32.json"},{"mnemonic":"v_cmp_class_f64","slug":"v_cmp_class_f64","records":1,"summary":"Evaluate the IEEE numeric class function specified as a 10 bit mask in the second input on the first input, a double-precision float, and set the…","page":"https://instructionsets.com/amdgpu/v_cmp_class_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_class_f64.json"},{"mnemonic":"v_cmp_eq_f16","slug":"v_cmp_eq_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_eq_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_eq_f16.json"},{"mnemonic":"v_cmp_eq_f32","slug":"v_cmp_eq_f32","records":1,"summary":"Per-lane single-precision equality compare, result written as an execution-mask-width bitmask.","page":"https://instructionsets.com/amdgpu/v_cmp_eq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_eq_f32.json"},{"mnemonic":"v_cmp_eq_f64","slug":"v_cmp_eq_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_eq_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_eq_f64.json"},{"mnemonic":"v_cmp_eq_i16","slug":"v_cmp_eq_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_eq_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_eq_i16.json"},{"mnemonic":"v_cmp_eq_i32","slug":"v_cmp_eq_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_eq_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_eq_i32.json"},{"mnemonic":"v_cmp_eq_i64","slug":"v_cmp_eq_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_eq_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_eq_i64.json"},{"mnemonic":"v_cmp_eq_u16","slug":"v_cmp_eq_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_eq_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_eq_u16.json"},{"mnemonic":"v_cmp_eq_u32","slug":"v_cmp_eq_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_eq_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_eq_u32.json"},{"mnemonic":"v_cmp_eq_u64","slug":"v_cmp_eq_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_eq_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_eq_u64.json"},{"mnemonic":"v_cmp_f_f16","slug":"v_cmp_f_f16","records":1,"summary":"Set the per-lane condition code to 0. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_f_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_f_f16.json"},{"mnemonic":"v_cmp_f_f32","slug":"v_cmp_f_f32","records":1,"summary":"Set the per-lane condition code to 0. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_f_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_f_f32.json"},{"mnemonic":"v_cmp_f_f64","slug":"v_cmp_f_f64","records":1,"summary":"Set the per-lane condition code to 0. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_f_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_f_f64.json"},{"mnemonic":"v_cmp_f_i16","slug":"v_cmp_f_i16","records":1,"summary":"Set the per-lane condition code to 0. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_f_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_f_i16.json"},{"mnemonic":"v_cmp_f_i32","slug":"v_cmp_f_i32","records":1,"summary":"Set the per-lane condition code to 0. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_f_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_f_i32.json"},{"mnemonic":"v_cmp_f_i64","slug":"v_cmp_f_i64","records":1,"summary":"Set the per-lane condition code to 0. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_f_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_f_i64.json"},{"mnemonic":"v_cmp_f_u16","slug":"v_cmp_f_u16","records":1,"summary":"Set the per-lane condition code to 0. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_f_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_f_u16.json"},{"mnemonic":"v_cmp_f_u32","slug":"v_cmp_f_u32","records":1,"summary":"Set the per-lane condition code to 0. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_f_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_f_u32.json"},{"mnemonic":"v_cmp_f_u64","slug":"v_cmp_f_u64","records":1,"summary":"Set the per-lane condition code to 0. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_f_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_f_u64.json"},{"mnemonic":"v_cmp_ge_f16","slug":"v_cmp_ge_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ge_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ge_f16.json"},{"mnemonic":"v_cmp_ge_f32","slug":"v_cmp_ge_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ge_f32.json"},{"mnemonic":"v_cmp_ge_f64","slug":"v_cmp_ge_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ge_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ge_f64.json"},{"mnemonic":"v_cmp_ge_i16","slug":"v_cmp_ge_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ge_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ge_i16.json"},{"mnemonic":"v_cmp_ge_i32","slug":"v_cmp_ge_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ge_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ge_i32.json"},{"mnemonic":"v_cmp_ge_i64","slug":"v_cmp_ge_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ge_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ge_i64.json"},{"mnemonic":"v_cmp_ge_u16","slug":"v_cmp_ge_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ge_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ge_u16.json"},{"mnemonic":"v_cmp_ge_u32","slug":"v_cmp_ge_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ge_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ge_u32.json"},{"mnemonic":"v_cmp_ge_u64","slug":"v_cmp_ge_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ge_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ge_u64.json"},{"mnemonic":"v_cmp_gt_f16","slug":"v_cmp_gt_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_gt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_gt_f16.json"},{"mnemonic":"v_cmp_gt_f32","slug":"v_cmp_gt_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_gt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_gt_f32.json"},{"mnemonic":"v_cmp_gt_f64","slug":"v_cmp_gt_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_gt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_gt_f64.json"},{"mnemonic":"v_cmp_gt_i16","slug":"v_cmp_gt_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_gt_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_gt_i16.json"},{"mnemonic":"v_cmp_gt_i32","slug":"v_cmp_gt_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_gt_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_gt_i32.json"},{"mnemonic":"v_cmp_gt_i64","slug":"v_cmp_gt_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_gt_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_gt_i64.json"},{"mnemonic":"v_cmp_gt_u16","slug":"v_cmp_gt_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_gt_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_gt_u16.json"},{"mnemonic":"v_cmp_gt_u32","slug":"v_cmp_gt_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_gt_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_gt_u32.json"},{"mnemonic":"v_cmp_gt_u64","slug":"v_cmp_gt_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_gt_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_gt_u64.json"},{"mnemonic":"v_cmp_le_f16","slug":"v_cmp_le_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_le_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_le_f16.json"},{"mnemonic":"v_cmp_le_f32","slug":"v_cmp_le_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_le_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_le_f32.json"},{"mnemonic":"v_cmp_le_f64","slug":"v_cmp_le_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_le_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_le_f64.json"},{"mnemonic":"v_cmp_le_i16","slug":"v_cmp_le_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_le_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_le_i16.json"},{"mnemonic":"v_cmp_le_i32","slug":"v_cmp_le_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_le_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_le_i32.json"},{"mnemonic":"v_cmp_le_i64","slug":"v_cmp_le_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_le_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_le_i64.json"},{"mnemonic":"v_cmp_le_u16","slug":"v_cmp_le_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_le_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_le_u16.json"},{"mnemonic":"v_cmp_le_u32","slug":"v_cmp_le_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_le_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_le_u32.json"},{"mnemonic":"v_cmp_le_u64","slug":"v_cmp_le_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_le_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_le_u64.json"},{"mnemonic":"v_cmp_lg_f16","slug":"v_cmp_lg_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_lg_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lg_f16.json"},{"mnemonic":"v_cmp_lg_f32","slug":"v_cmp_lg_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_lg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lg_f32.json"},{"mnemonic":"v_cmp_lg_f64","slug":"v_cmp_lg_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_lg_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lg_f64.json"},{"mnemonic":"v_cmp_lt_f16","slug":"v_cmp_lt_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_lt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lt_f16.json"},{"mnemonic":"v_cmp_lt_f32","slug":"v_cmp_lt_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_lt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lt_f32.json"},{"mnemonic":"v_cmp_lt_f64","slug":"v_cmp_lt_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_lt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lt_f64.json"},{"mnemonic":"v_cmp_lt_i16","slug":"v_cmp_lt_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_lt_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lt_i16.json"},{"mnemonic":"v_cmp_lt_i32","slug":"v_cmp_lt_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_lt_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lt_i32.json"},{"mnemonic":"v_cmp_lt_i64","slug":"v_cmp_lt_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_lt_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lt_i64.json"},{"mnemonic":"v_cmp_lt_u16","slug":"v_cmp_lt_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_lt_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lt_u16.json"},{"mnemonic":"v_cmp_lt_u32","slug":"v_cmp_lt_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_lt_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lt_u32.json"},{"mnemonic":"v_cmp_lt_u64","slug":"v_cmp_lt_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_lt_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_lt_u64.json"},{"mnemonic":"v_cmp_ne_i16","slug":"v_cmp_ne_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ne_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ne_i16.json"},{"mnemonic":"v_cmp_ne_i32","slug":"v_cmp_ne_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ne_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ne_i32.json"},{"mnemonic":"v_cmp_ne_i64","slug":"v_cmp_ne_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ne_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ne_i64.json"},{"mnemonic":"v_cmp_ne_u16","slug":"v_cmp_ne_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ne_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ne_u16.json"},{"mnemonic":"v_cmp_ne_u32","slug":"v_cmp_ne_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ne_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ne_u32.json"},{"mnemonic":"v_cmp_ne_u64","slug":"v_cmp_ne_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ne_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ne_u64.json"},{"mnemonic":"v_cmp_neq_f16","slug":"v_cmp_neq_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_neq_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_neq_f16.json"},{"mnemonic":"v_cmp_neq_f32","slug":"v_cmp_neq_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_neq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_neq_f32.json"},{"mnemonic":"v_cmp_neq_f64","slug":"v_cmp_neq_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_neq_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_neq_f64.json"},{"mnemonic":"v_cmp_nge_f16","slug":"v_cmp_nge_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_nge_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nge_f16.json"},{"mnemonic":"v_cmp_nge_f32","slug":"v_cmp_nge_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_nge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nge_f32.json"},{"mnemonic":"v_cmp_nge_f64","slug":"v_cmp_nge_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_nge_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nge_f64.json"},{"mnemonic":"v_cmp_ngt_f16","slug":"v_cmp_ngt_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ngt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ngt_f16.json"},{"mnemonic":"v_cmp_ngt_f32","slug":"v_cmp_ngt_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ngt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ngt_f32.json"},{"mnemonic":"v_cmp_ngt_f64","slug":"v_cmp_ngt_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_ngt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_ngt_f64.json"},{"mnemonic":"v_cmp_nle_f16","slug":"v_cmp_nle_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_nle_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nle_f16.json"},{"mnemonic":"v_cmp_nle_f32","slug":"v_cmp_nle_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_nle_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nle_f32.json"},{"mnemonic":"v_cmp_nle_f64","slug":"v_cmp_nle_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_nle_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nle_f64.json"},{"mnemonic":"v_cmp_nlg_f16","slug":"v_cmp_nlg_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_nlg_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nlg_f16.json"},{"mnemonic":"v_cmp_nlg_f32","slug":"v_cmp_nlg_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_nlg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nlg_f32.json"},{"mnemonic":"v_cmp_nlg_f64","slug":"v_cmp_nlg_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmp_nlg_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nlg_f64.json"},{"mnemonic":"v_cmp_nlt_f16","slug":"v_cmp_nlt_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_nlt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nlt_f16.json"},{"mnemonic":"v_cmp_nlt_f32","slug":"v_cmp_nlt_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_nlt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nlt_f32.json"},{"mnemonic":"v_cmp_nlt_f64","slug":"v_cmp_nlt_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_nlt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_nlt_f64.json"},{"mnemonic":"v_cmp_o_f16","slug":"v_cmp_o_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is orderable to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_o_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_o_f16.json"},{"mnemonic":"v_cmp_o_f32","slug":"v_cmp_o_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is orderable to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_o_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_o_f32.json"},{"mnemonic":"v_cmp_o_f64","slug":"v_cmp_o_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is orderable to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_o_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_o_f64.json"},{"mnemonic":"v_cmp_t_f16","slug":"v_cmp_t_f16","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_t_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_t_f16.json","aliases":["v_cmp_tru_f16"]},{"mnemonic":"v_cmp_t_f32","slug":"v_cmp_t_f32","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_t_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_t_f32.json","aliases":["v_cmp_tru_f32"]},{"mnemonic":"v_cmp_t_f64","slug":"v_cmp_t_f64","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_t_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_t_f64.json","aliases":["v_cmp_tru_f64"]},{"mnemonic":"v_cmp_t_i16","slug":"v_cmp_t_i16","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_t_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_t_i16.json"},{"mnemonic":"v_cmp_t_i32","slug":"v_cmp_t_i32","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_t_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_t_i32.json","aliases":["v_cmp_tru_i32"]},{"mnemonic":"v_cmp_t_i64","slug":"v_cmp_t_i64","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_t_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_t_i64.json","aliases":["v_cmp_tru_i64"]},{"mnemonic":"v_cmp_t_u16","slug":"v_cmp_t_u16","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_t_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_t_u16.json"},{"mnemonic":"v_cmp_t_u32","slug":"v_cmp_t_u32","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_t_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_t_u32.json","aliases":["v_cmp_tru_u32"]},{"mnemonic":"v_cmp_t_u64","slug":"v_cmp_t_u64","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_t_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_t_u64.json","aliases":["v_cmp_tru_u64"]},{"mnemonic":"v_cmp_tru_f16","slug":"v_cmp_tru_f16","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_tru_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_tru_f16.json","aliases":["v_cmp_t_f16"]},{"mnemonic":"v_cmp_tru_f32","slug":"v_cmp_tru_f32","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_tru_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_tru_f32.json","aliases":["v_cmp_t_f32"]},{"mnemonic":"v_cmp_tru_f64","slug":"v_cmp_tru_f64","records":1,"summary":"Set the per-lane condition code to 1. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_tru_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_tru_f64.json","aliases":["v_cmp_t_f64"]},{"mnemonic":"v_cmp_u_f16","slug":"v_cmp_u_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not orderable to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_u_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_u_f16.json"},{"mnemonic":"v_cmp_u_f32","slug":"v_cmp_u_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not orderable to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_u_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_u_f32.json"},{"mnemonic":"v_cmp_u_f64","slug":"v_cmp_u_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not orderable to the second input. Store the result into VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmp_u_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmp_u_f64.json"},{"mnemonic":"v_cmps_eq_f32","slug":"v_cmps_eq_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_eq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_eq_f32.json"},{"mnemonic":"v_cmps_eq_f64","slug":"v_cmps_eq_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_eq_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_eq_f64.json"},{"mnemonic":"v_cmps_f_f32","slug":"v_cmps_f_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_f_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_f_f32.json"},{"mnemonic":"v_cmps_f_f64","slug":"v_cmps_f_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_f_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_f_f64.json"},{"mnemonic":"v_cmps_ge_f32","slug":"v_cmps_ge_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_ge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_ge_f32.json"},{"mnemonic":"v_cmps_ge_f64","slug":"v_cmps_ge_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_ge_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_ge_f64.json"},{"mnemonic":"v_cmps_gt_f32","slug":"v_cmps_gt_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_gt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_gt_f32.json"},{"mnemonic":"v_cmps_gt_f64","slug":"v_cmps_gt_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_gt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_gt_f64.json"},{"mnemonic":"v_cmps_le_f32","slug":"v_cmps_le_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_le_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_le_f32.json"},{"mnemonic":"v_cmps_le_f64","slug":"v_cmps_le_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_le_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_le_f64.json"},{"mnemonic":"v_cmps_lg_f32","slug":"v_cmps_lg_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_lg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_lg_f32.json"},{"mnemonic":"v_cmps_lg_f64","slug":"v_cmps_lg_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_lg_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_lg_f64.json"},{"mnemonic":"v_cmps_lt_f32","slug":"v_cmps_lt_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_lt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_lt_f32.json"},{"mnemonic":"v_cmps_lt_f64","slug":"v_cmps_lt_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_lt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_lt_f64.json"},{"mnemonic":"v_cmps_neq_f32","slug":"v_cmps_neq_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_neq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_neq_f32.json"},{"mnemonic":"v_cmps_neq_f64","slug":"v_cmps_neq_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_neq_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_neq_f64.json"},{"mnemonic":"v_cmps_nge_f32","slug":"v_cmps_nge_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_nge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_nge_f32.json"},{"mnemonic":"v_cmps_nge_f64","slug":"v_cmps_nge_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_nge_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_nge_f64.json"},{"mnemonic":"v_cmps_ngt_f32","slug":"v_cmps_ngt_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_ngt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_ngt_f32.json"},{"mnemonic":"v_cmps_ngt_f64","slug":"v_cmps_ngt_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_ngt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_ngt_f64.json"},{"mnemonic":"v_cmps_nle_f32","slug":"v_cmps_nle_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_nle_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_nle_f32.json"},{"mnemonic":"v_cmps_nle_f64","slug":"v_cmps_nle_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_nle_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_nle_f64.json"},{"mnemonic":"v_cmps_nlg_f32","slug":"v_cmps_nlg_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_nlg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_nlg_f32.json"},{"mnemonic":"v_cmps_nlg_f64","slug":"v_cmps_nlg_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_nlg_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_nlg_f64.json"},{"mnemonic":"v_cmps_nlt_f32","slug":"v_cmps_nlt_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_nlt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_nlt_f32.json"},{"mnemonic":"v_cmps_nlt_f64","slug":"v_cmps_nlt_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_nlt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_nlt_f64.json"},{"mnemonic":"v_cmps_o_f32","slug":"v_cmps_o_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_o_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_o_f32.json"},{"mnemonic":"v_cmps_o_f64","slug":"v_cmps_o_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_o_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_o_f64.json"},{"mnemonic":"v_cmps_tru_f32","slug":"v_cmps_tru_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_tru_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_tru_f32.json"},{"mnemonic":"v_cmps_tru_f64","slug":"v_cmps_tru_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_tru_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_tru_f64.json"},{"mnemonic":"v_cmps_u_f32","slug":"v_cmps_u_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_u_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_u_f32.json"},{"mnemonic":"v_cmps_u_f64","slug":"v_cmps_u_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmps_u_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmps_u_f64.json"},{"mnemonic":"v_cmpsx_eq_f32","slug":"v_cmpsx_eq_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_eq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_eq_f32.json"},{"mnemonic":"v_cmpsx_eq_f64","slug":"v_cmpsx_eq_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_eq_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_eq_f64.json"},{"mnemonic":"v_cmpsx_f_f32","slug":"v_cmpsx_f_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_f_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_f_f32.json"},{"mnemonic":"v_cmpsx_f_f64","slug":"v_cmpsx_f_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_f_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_f_f64.json"},{"mnemonic":"v_cmpsx_ge_f32","slug":"v_cmpsx_ge_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_ge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_ge_f32.json"},{"mnemonic":"v_cmpsx_ge_f64","slug":"v_cmpsx_ge_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_ge_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_ge_f64.json"},{"mnemonic":"v_cmpsx_gt_f32","slug":"v_cmpsx_gt_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_gt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_gt_f32.json"},{"mnemonic":"v_cmpsx_gt_f64","slug":"v_cmpsx_gt_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_gt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_gt_f64.json"},{"mnemonic":"v_cmpsx_le_f32","slug":"v_cmpsx_le_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_le_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_le_f32.json"},{"mnemonic":"v_cmpsx_le_f64","slug":"v_cmpsx_le_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_le_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_le_f64.json"},{"mnemonic":"v_cmpsx_lg_f32","slug":"v_cmpsx_lg_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_lg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_lg_f32.json"},{"mnemonic":"v_cmpsx_lg_f64","slug":"v_cmpsx_lg_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_lg_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_lg_f64.json"},{"mnemonic":"v_cmpsx_lt_f32","slug":"v_cmpsx_lt_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_lt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_lt_f32.json"},{"mnemonic":"v_cmpsx_lt_f64","slug":"v_cmpsx_lt_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_lt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_lt_f64.json"},{"mnemonic":"v_cmpsx_neq_f32","slug":"v_cmpsx_neq_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_neq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_neq_f32.json"},{"mnemonic":"v_cmpsx_neq_f64","slug":"v_cmpsx_neq_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_neq_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_neq_f64.json"},{"mnemonic":"v_cmpsx_nge_f32","slug":"v_cmpsx_nge_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_nge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_nge_f32.json"},{"mnemonic":"v_cmpsx_nge_f64","slug":"v_cmpsx_nge_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_nge_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_nge_f64.json"},{"mnemonic":"v_cmpsx_ngt_f32","slug":"v_cmpsx_ngt_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_ngt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_ngt_f32.json"},{"mnemonic":"v_cmpsx_ngt_f64","slug":"v_cmpsx_ngt_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_ngt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_ngt_f64.json"},{"mnemonic":"v_cmpsx_nle_f32","slug":"v_cmpsx_nle_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_nle_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_nle_f32.json"},{"mnemonic":"v_cmpsx_nle_f64","slug":"v_cmpsx_nle_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_nle_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_nle_f64.json"},{"mnemonic":"v_cmpsx_nlg_f32","slug":"v_cmpsx_nlg_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_nlg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_nlg_f32.json"},{"mnemonic":"v_cmpsx_nlg_f64","slug":"v_cmpsx_nlg_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_nlg_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_nlg_f64.json"},{"mnemonic":"v_cmpsx_nlt_f32","slug":"v_cmpsx_nlt_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_nlt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_nlt_f32.json"},{"mnemonic":"v_cmpsx_nlt_f64","slug":"v_cmpsx_nlt_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_nlt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_nlt_f64.json"},{"mnemonic":"v_cmpsx_o_f32","slug":"v_cmpsx_o_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_o_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_o_f32.json"},{"mnemonic":"v_cmpsx_o_f64","slug":"v_cmpsx_o_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_o_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_o_f64.json"},{"mnemonic":"v_cmpsx_tru_f32","slug":"v_cmpsx_tru_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_tru_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_tru_f32.json"},{"mnemonic":"v_cmpsx_tru_f64","slug":"v_cmpsx_tru_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_tru_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_tru_f64.json"},{"mnemonic":"v_cmpsx_u_f32","slug":"v_cmpsx_u_f32","records":1,"summary":"AMDGPU VOPC vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_u_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_u_f32.json"},{"mnemonic":"v_cmpsx_u_f64","slug":"v_cmpsx_u_f64","records":1,"summary":"AMDGPU VOPC vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cmpsx_u_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpsx_u_f64.json"},{"mnemonic":"v_cmpx_class_f16","slug":"v_cmpx_class_f16","records":1,"summary":"Evaluate the IEEE numeric class function specified as a 10 bit mask in the second input on the first input, a half-precision float, and set the…","page":"https://instructionsets.com/amdgpu/v_cmpx_class_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_class_f16.json"},{"mnemonic":"v_cmpx_class_f32","slug":"v_cmpx_class_f32","records":1,"summary":"Evaluate the IEEE numeric class function specified as a 10 bit mask in the second input on the first input, a single-precision float, and set the…","page":"https://instructionsets.com/amdgpu/v_cmpx_class_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_class_f32.json"},{"mnemonic":"v_cmpx_class_f64","slug":"v_cmpx_class_f64","records":1,"summary":"Evaluate the IEEE numeric class function specified as a 10 bit mask in the second input on the first input, a double-precision float, and set the…","page":"https://instructionsets.com/amdgpu/v_cmpx_class_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_class_f64.json"},{"mnemonic":"v_cmpx_eq_f16","slug":"v_cmpx_eq_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_eq_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_eq_f16.json"},{"mnemonic":"v_cmpx_eq_f32","slug":"v_cmpx_eq_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_eq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_eq_f32.json"},{"mnemonic":"v_cmpx_eq_f64","slug":"v_cmpx_eq_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_eq_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_eq_f64.json"},{"mnemonic":"v_cmpx_eq_i16","slug":"v_cmpx_eq_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_eq_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_eq_i16.json"},{"mnemonic":"v_cmpx_eq_i32","slug":"v_cmpx_eq_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_eq_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_eq_i32.json"},{"mnemonic":"v_cmpx_eq_i64","slug":"v_cmpx_eq_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_eq_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_eq_i64.json"},{"mnemonic":"v_cmpx_eq_u16","slug":"v_cmpx_eq_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_eq_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_eq_u16.json"},{"mnemonic":"v_cmpx_eq_u32","slug":"v_cmpx_eq_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_eq_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_eq_u32.json"},{"mnemonic":"v_cmpx_eq_u64","slug":"v_cmpx_eq_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_eq_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_eq_u64.json"},{"mnemonic":"v_cmpx_f_f16","slug":"v_cmpx_f_f16","records":1,"summary":"Set the per-lane condition code to 0. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_f_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_f_f16.json"},{"mnemonic":"v_cmpx_f_f32","slug":"v_cmpx_f_f32","records":1,"summary":"Set the per-lane condition code to 0. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_f_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_f_f32.json"},{"mnemonic":"v_cmpx_f_f64","slug":"v_cmpx_f_f64","records":1,"summary":"Set the per-lane condition code to 0. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_f_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_f_f64.json"},{"mnemonic":"v_cmpx_f_i16","slug":"v_cmpx_f_i16","records":1,"summary":"Set the per-lane condition code to 0. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_f_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_f_i16.json"},{"mnemonic":"v_cmpx_f_i32","slug":"v_cmpx_f_i32","records":1,"summary":"Set the per-lane condition code to 0. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_f_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_f_i32.json"},{"mnemonic":"v_cmpx_f_i64","slug":"v_cmpx_f_i64","records":1,"summary":"Set the per-lane condition code to 0. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_f_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_f_i64.json"},{"mnemonic":"v_cmpx_f_u16","slug":"v_cmpx_f_u16","records":1,"summary":"Set the per-lane condition code to 0. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_f_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_f_u16.json"},{"mnemonic":"v_cmpx_f_u32","slug":"v_cmpx_f_u32","records":1,"summary":"Set the per-lane condition code to 0. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_f_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_f_u32.json"},{"mnemonic":"v_cmpx_f_u64","slug":"v_cmpx_f_u64","records":1,"summary":"Set the per-lane condition code to 0. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_f_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_f_u64.json"},{"mnemonic":"v_cmpx_ge_f16","slug":"v_cmpx_ge_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ge_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ge_f16.json"},{"mnemonic":"v_cmpx_ge_f32","slug":"v_cmpx_ge_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ge_f32.json"},{"mnemonic":"v_cmpx_ge_f64","slug":"v_cmpx_ge_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ge_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ge_f64.json"},{"mnemonic":"v_cmpx_ge_i16","slug":"v_cmpx_ge_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ge_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ge_i16.json"},{"mnemonic":"v_cmpx_ge_i32","slug":"v_cmpx_ge_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ge_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ge_i32.json"},{"mnemonic":"v_cmpx_ge_i64","slug":"v_cmpx_ge_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ge_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ge_i64.json"},{"mnemonic":"v_cmpx_ge_u16","slug":"v_cmpx_ge_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ge_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ge_u16.json"},{"mnemonic":"v_cmpx_ge_u32","slug":"v_cmpx_ge_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ge_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ge_u32.json"},{"mnemonic":"v_cmpx_ge_u64","slug":"v_cmpx_ge_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ge_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ge_u64.json"},{"mnemonic":"v_cmpx_gt_f16","slug":"v_cmpx_gt_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_gt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_gt_f16.json"},{"mnemonic":"v_cmpx_gt_f32","slug":"v_cmpx_gt_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_gt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_gt_f32.json"},{"mnemonic":"v_cmpx_gt_f64","slug":"v_cmpx_gt_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_gt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_gt_f64.json"},{"mnemonic":"v_cmpx_gt_i16","slug":"v_cmpx_gt_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_gt_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_gt_i16.json"},{"mnemonic":"v_cmpx_gt_i32","slug":"v_cmpx_gt_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_gt_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_gt_i32.json"},{"mnemonic":"v_cmpx_gt_i64","slug":"v_cmpx_gt_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_gt_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_gt_i64.json"},{"mnemonic":"v_cmpx_gt_u16","slug":"v_cmpx_gt_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_gt_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_gt_u16.json"},{"mnemonic":"v_cmpx_gt_u32","slug":"v_cmpx_gt_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_gt_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_gt_u32.json"},{"mnemonic":"v_cmpx_gt_u64","slug":"v_cmpx_gt_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_gt_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_gt_u64.json"},{"mnemonic":"v_cmpx_le_f16","slug":"v_cmpx_le_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_le_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_le_f16.json"},{"mnemonic":"v_cmpx_le_f32","slug":"v_cmpx_le_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_le_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_le_f32.json"},{"mnemonic":"v_cmpx_le_f64","slug":"v_cmpx_le_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_le_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_le_f64.json"},{"mnemonic":"v_cmpx_le_i16","slug":"v_cmpx_le_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_le_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_le_i16.json"},{"mnemonic":"v_cmpx_le_i32","slug":"v_cmpx_le_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_le_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_le_i32.json"},{"mnemonic":"v_cmpx_le_i64","slug":"v_cmpx_le_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_le_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_le_i64.json"},{"mnemonic":"v_cmpx_le_u16","slug":"v_cmpx_le_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_le_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_le_u16.json"},{"mnemonic":"v_cmpx_le_u32","slug":"v_cmpx_le_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_le_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_le_u32.json"},{"mnemonic":"v_cmpx_le_u64","slug":"v_cmpx_le_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_le_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_le_u64.json"},{"mnemonic":"v_cmpx_lg_f16","slug":"v_cmpx_lg_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lg_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lg_f16.json"},{"mnemonic":"v_cmpx_lg_f32","slug":"v_cmpx_lg_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lg_f32.json"},{"mnemonic":"v_cmpx_lg_f64","slug":"v_cmpx_lg_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lg_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lg_f64.json"},{"mnemonic":"v_cmpx_lt_f16","slug":"v_cmpx_lt_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lt_f16.json"},{"mnemonic":"v_cmpx_lt_f32","slug":"v_cmpx_lt_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lt_f32.json"},{"mnemonic":"v_cmpx_lt_f64","slug":"v_cmpx_lt_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lt_f64.json"},{"mnemonic":"v_cmpx_lt_i16","slug":"v_cmpx_lt_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lt_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lt_i16.json"},{"mnemonic":"v_cmpx_lt_i32","slug":"v_cmpx_lt_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lt_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lt_i32.json"},{"mnemonic":"v_cmpx_lt_i64","slug":"v_cmpx_lt_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lt_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lt_i64.json"},{"mnemonic":"v_cmpx_lt_u16","slug":"v_cmpx_lt_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lt_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lt_u16.json"},{"mnemonic":"v_cmpx_lt_u32","slug":"v_cmpx_lt_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lt_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lt_u32.json"},{"mnemonic":"v_cmpx_lt_u64","slug":"v_cmpx_lt_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_lt_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_lt_u64.json"},{"mnemonic":"v_cmpx_ne_i16","slug":"v_cmpx_ne_i16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ne_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ne_i16.json"},{"mnemonic":"v_cmpx_ne_i32","slug":"v_cmpx_ne_i32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ne_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ne_i32.json"},{"mnemonic":"v_cmpx_ne_i64","slug":"v_cmpx_ne_i64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ne_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ne_i64.json"},{"mnemonic":"v_cmpx_ne_u16","slug":"v_cmpx_ne_u16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ne_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ne_u16.json"},{"mnemonic":"v_cmpx_ne_u32","slug":"v_cmpx_ne_u32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ne_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ne_u32.json"},{"mnemonic":"v_cmpx_ne_u64","slug":"v_cmpx_ne_u64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ne_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ne_u64.json"},{"mnemonic":"v_cmpx_neq_f16","slug":"v_cmpx_neq_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_neq_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_neq_f16.json"},{"mnemonic":"v_cmpx_neq_f32","slug":"v_cmpx_neq_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_neq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_neq_f32.json"},{"mnemonic":"v_cmpx_neq_f64","slug":"v_cmpx_neq_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_neq_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_neq_f64.json"},{"mnemonic":"v_cmpx_nge_f16","slug":"v_cmpx_nge_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nge_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nge_f16.json"},{"mnemonic":"v_cmpx_nge_f32","slug":"v_cmpx_nge_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nge_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nge_f32.json"},{"mnemonic":"v_cmpx_nge_f64","slug":"v_cmpx_nge_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nge_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nge_f64.json"},{"mnemonic":"v_cmpx_ngt_f16","slug":"v_cmpx_ngt_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ngt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ngt_f16.json"},{"mnemonic":"v_cmpx_ngt_f32","slug":"v_cmpx_ngt_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ngt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ngt_f32.json"},{"mnemonic":"v_cmpx_ngt_f64","slug":"v_cmpx_ngt_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_ngt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_ngt_f64.json"},{"mnemonic":"v_cmpx_nle_f16","slug":"v_cmpx_nle_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nle_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nle_f16.json"},{"mnemonic":"v_cmpx_nle_f32","slug":"v_cmpx_nle_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nle_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nle_f32.json"},{"mnemonic":"v_cmpx_nle_f64","slug":"v_cmpx_nle_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or equal to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nle_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nle_f64.json"},{"mnemonic":"v_cmpx_nlg_f16","slug":"v_cmpx_nlg_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nlg_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nlg_f16.json"},{"mnemonic":"v_cmpx_nlg_f32","slug":"v_cmpx_nlg_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nlg_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nlg_f32.json"},{"mnemonic":"v_cmpx_nlg_f64","slug":"v_cmpx_nlg_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than or greater than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nlg_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nlg_f64.json"},{"mnemonic":"v_cmpx_nlt_f16","slug":"v_cmpx_nlt_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nlt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nlt_f16.json"},{"mnemonic":"v_cmpx_nlt_f32","slug":"v_cmpx_nlt_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nlt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nlt_f32.json"},{"mnemonic":"v_cmpx_nlt_f64","slug":"v_cmpx_nlt_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not less than the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_nlt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_nlt_f64.json"},{"mnemonic":"v_cmpx_o_f16","slug":"v_cmpx_o_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is orderable to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_o_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_o_f16.json"},{"mnemonic":"v_cmpx_o_f32","slug":"v_cmpx_o_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is orderable to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_o_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_o_f32.json"},{"mnemonic":"v_cmpx_o_f64","slug":"v_cmpx_o_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is orderable to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_o_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_o_f64.json"},{"mnemonic":"v_cmpx_t_f16","slug":"v_cmpx_t_f16","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask.","page":"https://instructionsets.com/amdgpu/v_cmpx_t_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_t_f16.json","aliases":["v_cmpx_tru_f16"]},{"mnemonic":"v_cmpx_t_f32","slug":"v_cmpx_t_f32","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask.","page":"https://instructionsets.com/amdgpu/v_cmpx_t_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_t_f32.json","aliases":["v_cmpx_tru_f32"]},{"mnemonic":"v_cmpx_t_f64","slug":"v_cmpx_t_f64","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask.","page":"https://instructionsets.com/amdgpu/v_cmpx_t_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_t_f64.json","aliases":["v_cmpx_tru_f64"]},{"mnemonic":"v_cmpx_t_i16","slug":"v_cmpx_t_i16","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_t_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_t_i16.json"},{"mnemonic":"v_cmpx_t_i32","slug":"v_cmpx_t_i32","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_t_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_t_i32.json","aliases":["v_cmpx_tru_i32"]},{"mnemonic":"v_cmpx_t_i64","slug":"v_cmpx_t_i64","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_t_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_t_i64.json","aliases":["v_cmpx_tru_i64"]},{"mnemonic":"v_cmpx_t_u16","slug":"v_cmpx_t_u16","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_t_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_t_u16.json"},{"mnemonic":"v_cmpx_t_u32","slug":"v_cmpx_t_u32","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_t_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_t_u32.json","aliases":["v_cmpx_tru_u32"]},{"mnemonic":"v_cmpx_t_u64","slug":"v_cmpx_t_u64","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_t_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_t_u64.json","aliases":["v_cmpx_tru_u64"]},{"mnemonic":"v_cmpx_tru_f16","slug":"v_cmpx_tru_f16","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_tru_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_tru_f16.json","aliases":["v_cmpx_t_f16"]},{"mnemonic":"v_cmpx_tru_f32","slug":"v_cmpx_tru_f32","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_tru_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_tru_f32.json","aliases":["v_cmpx_t_f32"]},{"mnemonic":"v_cmpx_tru_f64","slug":"v_cmpx_tru_f64","records":1,"summary":"Set the per-lane condition code to 1. Store the result into the EXEC mask and to VCC or a scalar register.","page":"https://instructionsets.com/amdgpu/v_cmpx_tru_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_tru_f64.json","aliases":["v_cmpx_t_f64"]},{"mnemonic":"v_cmpx_u_f16","slug":"v_cmpx_u_f16","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not orderable to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_u_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_u_f16.json"},{"mnemonic":"v_cmpx_u_f32","slug":"v_cmpx_u_f32","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not orderable to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_u_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_u_f32.json"},{"mnemonic":"v_cmpx_u_f64","slug":"v_cmpx_u_f64","records":1,"summary":"Set the per-lane condition code to 1 iff the first input is not orderable to the second input.","page":"https://instructionsets.com/amdgpu/v_cmpx_u_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cmpx_u_f64.json"},{"mnemonic":"v_cndmask_b16","slug":"v_cndmask_b16","records":1,"summary":"Copy data from one of two inputs based on the per-lane condition code and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cndmask_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cndmask_b16.json"},{"mnemonic":"v_cndmask_b16_fake16","slug":"v_cndmask_b16_fake16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cndmask_b16_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cndmask_b16_fake16.json"},{"mnemonic":"v_cndmask_b16_t16","slug":"v_cndmask_b16_t16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cndmask_b16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cndmask_b16_t16.json"},{"mnemonic":"v_cndmask_b32","slug":"v_cndmask_b32","records":1,"summary":"Copy data from one of two inputs based on the per-lane condition code and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cndmask_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cndmask_b32.json"},{"mnemonic":"v_cos_bf16","slug":"v_cos_bf16","records":1,"summary":"AMDGPU VOP1 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cos_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cos_bf16.json"},{"mnemonic":"v_cos_f16","slug":"v_cos_f16","records":1,"summary":"Calculate the trigonometric cosine of a half-precision float value using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cos_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cos_f16.json"},{"mnemonic":"v_cos_f32","slug":"v_cos_f32","records":1,"summary":"Calculate the trigonometric cosine of a single-precision float value using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cos_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cos_f32.json"},{"mnemonic":"v_cubeid_f32","slug":"v_cubeid_f32","records":1,"summary":"Compute the cubemap face ID of a 3D coordinate specified as three single-precision float inputs.","page":"https://instructionsets.com/amdgpu/v_cubeid_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cubeid_f32.json"},{"mnemonic":"v_cubema_f32","slug":"v_cubema_f32","records":1,"summary":"Compute the cubemap major axis of a 3D coordinate specified as three single-precision float inputs.","page":"https://instructionsets.com/amdgpu/v_cubema_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cubema_f32.json"},{"mnemonic":"v_cubesc_f32","slug":"v_cubesc_f32","records":1,"summary":"Compute the cubemap S coordinate of a 3D coordinate specified as three single-precision float inputs.","page":"https://instructionsets.com/amdgpu/v_cubesc_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cubesc_f32.json"},{"mnemonic":"v_cubetc_f32","slug":"v_cubetc_f32","records":1,"summary":"Compute the cubemap T coordinate of a 3D coordinate specified as three single-precision float inputs.","page":"https://instructionsets.com/amdgpu/v_cubetc_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cubetc_f32.json"},{"mnemonic":"v_cvt_f16_bf8","slug":"v_cvt_f16_bf8","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_f16_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f16_bf8.json"},{"mnemonic":"v_cvt_f16_f32","slug":"v_cvt_f16_f32","records":1,"summary":"Convert from a single-precision float input to a half-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f16_f32.json"},{"mnemonic":"v_cvt_f16_f32_fake16","slug":"v_cvt_f16_f32_fake16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16/f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_f16_f32_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f16_f32_fake16.json"},{"mnemonic":"v_cvt_f16_f32_t16","slug":"v_cvt_f16_f32_t16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16/f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_f16_f32_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f16_f32_t16.json"},{"mnemonic":"v_cvt_f16_fp8","slug":"v_cvt_f16_fp8","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_f16_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f16_fp8.json"},{"mnemonic":"v_cvt_f16_i16","slug":"v_cvt_f16_i16","records":1,"summary":"Convert from a signed 16-bit integer input to a half-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f16_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f16_i16.json"},{"mnemonic":"v_cvt_f16_u16","slug":"v_cvt_f16_u16","records":1,"summary":"Convert from an unsigned 16-bit integer input to a half-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f16_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f16_u16.json"},{"mnemonic":"v_cvt_f32_bf16","slug":"v_cvt_f32_bf16","records":1,"summary":"Convert from a BF16 float input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_bf16.json"},{"mnemonic":"v_cvt_f32_bf8","slug":"v_cvt_f32_bf8","records":1,"summary":"Convert from a BF8 float input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_bf8.json"},{"mnemonic":"v_cvt_f32_bf8_op_sel","slug":"v_cvt_f32_bf8_op_sel","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_f32_bf8_op_sel/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_bf8_op_sel.json"},{"mnemonic":"v_cvt_f32_f16","slug":"v_cvt_f32_f16","records":1,"summary":"Convert from a half-precision float input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_f16.json"},{"mnemonic":"v_cvt_f32_f16_fake16","slug":"v_cvt_f32_f16_fake16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16/f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_f32_f16_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_f16_fake16.json"},{"mnemonic":"v_cvt_f32_f16_t16","slug":"v_cvt_f32_f16_t16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16/f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_f32_f16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_f16_t16.json"},{"mnemonic":"v_cvt_f32_f64","slug":"v_cvt_f32_f64","records":1,"summary":"Convert from a double-precision float input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_f64.json"},{"mnemonic":"v_cvt_f32_fp8","slug":"v_cvt_f32_fp8","records":1,"summary":"Convert from an FP8 float input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_fp8.json"},{"mnemonic":"v_cvt_f32_fp8_gfx1250","slug":"v_cvt_f32_fp8_gfx1250","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_f32_fp8_gfx1250/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_fp8_gfx1250.json"},{"mnemonic":"v_cvt_f32_fp8_op_sel","slug":"v_cvt_f32_fp8_op_sel","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_f32_fp8_op_sel/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_fp8_op_sel.json"},{"mnemonic":"v_cvt_f32_i32","slug":"v_cvt_f32_i32","records":1,"summary":"Per-lane conversion from signed 32-bit integer to single-precision float.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_i32.json"},{"mnemonic":"v_cvt_f32_u32","slug":"v_cvt_f32_u32","records":1,"summary":"Convert from an unsigned 32-bit integer input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_u32.json"},{"mnemonic":"v_cvt_f32_ubyte0","slug":"v_cvt_f32_ubyte0","records":1,"summary":"Convert an unsigned byte in byte 0 of the input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_ubyte0/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_ubyte0.json"},{"mnemonic":"v_cvt_f32_ubyte1","slug":"v_cvt_f32_ubyte1","records":1,"summary":"Convert an unsigned byte in byte 1 of the input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_ubyte1/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_ubyte1.json"},{"mnemonic":"v_cvt_f32_ubyte2","slug":"v_cvt_f32_ubyte2","records":1,"summary":"Convert an unsigned byte in byte 2 of the input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_ubyte2/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_ubyte2.json"},{"mnemonic":"v_cvt_f32_ubyte3","slug":"v_cvt_f32_ubyte3","records":1,"summary":"Convert an unsigned byte in byte 3 of the input to a single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f32_ubyte3/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f32_ubyte3.json"},{"mnemonic":"v_cvt_f64_f32","slug":"v_cvt_f64_f32","records":1,"summary":"Convert from a single-precision float input to a double-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f64_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f64_f32.json"},{"mnemonic":"v_cvt_f64_i32","slug":"v_cvt_f64_i32","records":1,"summary":"Convert from a signed 32-bit integer input to a double-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f64_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f64_i32.json"},{"mnemonic":"v_cvt_f64_u32","slug":"v_cvt_f64_u32","records":1,"summary":"Convert from an unsigned 32-bit integer input to a double-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_f64_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_f64_u32.json"},{"mnemonic":"v_cvt_flr_i32_f32","slug":"v_cvt_flr_i32_f32","records":1,"summary":"Convert from a single-precision float input to a signed 32-bit integer value using round-down semantics (ignore the default rounding mode) and store…","page":"https://instructionsets.com/amdgpu/v_cvt_flr_i32_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_flr_i32_f32.json","aliases":["v_cvt_floor_i32_f32"]},{"mnemonic":"v_cvt_i16_f16","slug":"v_cvt_i16_f16","records":1,"summary":"Convert from a half-precision float input to a signed 16-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_i16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_i16_f16.json"},{"mnemonic":"v_cvt_i32_f32","slug":"v_cvt_i32_f32","records":1,"summary":"Convert from a single-precision float input to a signed 32-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_i32_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_i32_f32.json"},{"mnemonic":"v_cvt_i32_f64","slug":"v_cvt_i32_f64","records":1,"summary":"Convert from a double-precision float input to a signed 32-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_i32_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_i32_f64.json"},{"mnemonic":"v_cvt_i32_i16","slug":"v_cvt_i32_i16","records":1,"summary":"Convert from a signed 16-bit integer input to a signed 32-bit integer value using sign extension and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_i32_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_i32_i16.json"},{"mnemonic":"v_cvt_norm_i16_f16","slug":"v_cvt_norm_i16_f16","records":1,"summary":"Convert from a half-precision float input to a signed normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_norm_i16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_norm_i16_f16.json"},{"mnemonic":"v_cvt_norm_u16_f16","slug":"v_cvt_norm_u16_f16","records":1,"summary":"Convert from a half-precision float input to an unsigned normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_norm_u16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_norm_u16_f16.json"},{"mnemonic":"v_cvt_off_f32_i4","slug":"v_cvt_off_f32_i4","records":1,"summary":"Convert from a signed 4-bit integer input to a single-precision float value using an offset table and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_off_f32_i4/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_off_f32_i4.json"},{"mnemonic":"v_cvt_pk_bf16_f32","slug":"v_cvt_pk_bf16_f32","records":1,"summary":"Convert from two single-precision float inputs to a packed BF16 value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_bf16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_bf16_f32.json"},{"mnemonic":"v_cvt_pk_bf8_f16","slug":"v_cvt_pk_bf8_f16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_pk_bf8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_bf8_f16.json"},{"mnemonic":"v_cvt_pk_bf8_f32","slug":"v_cvt_pk_bf8_f32","records":1,"summary":"Convert from two single-precision float inputs to a packed BF8 float value with round to nearest even semantics and store the result into 16 bits of…","page":"https://instructionsets.com/amdgpu/v_cvt_pk_bf8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_bf8_f32.json"},{"mnemonic":"v_cvt_pk_f16_bf8","slug":"v_cvt_pk_f16_bf8","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_pk_f16_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_f16_bf8.json"},{"mnemonic":"v_cvt_pk_f16_f32","slug":"v_cvt_pk_f16_f32","records":1,"summary":"Convert from two single-precision float inputs to a packed half-precision value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_f16_f32.json"},{"mnemonic":"v_cvt_pk_f16_fp8","slug":"v_cvt_pk_f16_fp8","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_pk_f16_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_f16_fp8.json"},{"mnemonic":"v_cvt_pk_f32_bf8","slug":"v_cvt_pk_f32_bf8","records":1,"summary":"Convert from a packed 2-component BF8 float input to a packed single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_f32_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_f32_bf8.json"},{"mnemonic":"v_cvt_pk_f32_bf8_fake16","slug":"v_cvt_pk_f32_bf8_fake16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_pk_f32_bf8_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_f32_bf8_fake16.json"},{"mnemonic":"v_cvt_pk_f32_bf8_t16","slug":"v_cvt_pk_f32_bf8_t16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_pk_f32_bf8_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_f32_bf8_t16.json"},{"mnemonic":"v_cvt_pk_f32_fp8","slug":"v_cvt_pk_f32_fp8","records":1,"summary":"Convert from a packed 2-component FP8 float input to a packed single-precision float value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_f32_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_f32_fp8.json"},{"mnemonic":"v_cvt_pk_f32_fp8_fake16","slug":"v_cvt_pk_f32_fp8_fake16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_pk_f32_fp8_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_f32_fp8_fake16.json"},{"mnemonic":"v_cvt_pk_f32_fp8_t16","slug":"v_cvt_pk_f32_fp8_t16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_pk_f32_fp8_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_f32_fp8_t16.json"},{"mnemonic":"v_cvt_pk_fp8_f16","slug":"v_cvt_pk_fp8_f16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_pk_fp8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_fp8_f16.json"},{"mnemonic":"v_cvt_pk_fp8_f32","slug":"v_cvt_pk_fp8_f32","records":1,"summary":"Convert from two single-precision float inputs to a packed FP8 float value with round to nearest even semantics and store the result into 16 bits of…","page":"https://instructionsets.com/amdgpu/v_cvt_pk_fp8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_fp8_f32.json"},{"mnemonic":"v_cvt_pk_fp8_f32_gfx1250","slug":"v_cvt_pk_fp8_f32_gfx1250","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_pk_fp8_f32_gfx1250/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_fp8_f32_gfx1250.json"},{"mnemonic":"v_cvt_pk_i16_f32","slug":"v_cvt_pk_i16_f32","records":1,"summary":"Convert two single-precision float inputs into a packed signed 16-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_i16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_i16_f32.json"},{"mnemonic":"v_cvt_pk_i16_i32","slug":"v_cvt_pk_i16_i32","records":1,"summary":"Convert from two signed 32-bit integer inputs to a packed signed 16-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_i16_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_i16_i32.json"},{"mnemonic":"v_cvt_pk_norm_i16_f16","slug":"v_cvt_pk_norm_i16_f16","records":1,"summary":"Convert from two half-precision float inputs to a packed signed normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_norm_i16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_norm_i16_f16.json","aliases":["v_cvt_pknorm_i16_f16"]},{"mnemonic":"v_cvt_pk_norm_i16_f32","slug":"v_cvt_pk_norm_i16_f32","records":1,"summary":"Convert from two single-precision float inputs to a packed signed normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_norm_i16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_norm_i16_f32.json","aliases":["v_cvt_pknorm_i16_f32"]},{"mnemonic":"v_cvt_pk_norm_u16_f16","slug":"v_cvt_pk_norm_u16_f16","records":1,"summary":"Convert from two half-precision float inputs to a packed unsigned normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_norm_u16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_norm_u16_f16.json","aliases":["v_cvt_pknorm_u16_f16"]},{"mnemonic":"v_cvt_pk_norm_u16_f32","slug":"v_cvt_pk_norm_u16_f32","records":1,"summary":"Convert from two single-precision float inputs to a packed unsigned normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_norm_u16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_norm_u16_f32.json","aliases":["v_cvt_pknorm_u16_f32"]},{"mnemonic":"v_cvt_pk_u16_f32","slug":"v_cvt_pk_u16_f32","records":1,"summary":"Convert two single-precision float inputs into a packed unsigned 16-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_u16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_u16_f32.json"},{"mnemonic":"v_cvt_pk_u16_u32","slug":"v_cvt_pk_u16_u32","records":1,"summary":"Convert from two unsigned 32-bit integer inputs to a packed unsigned 16-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pk_u16_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_u16_u32.json"},{"mnemonic":"v_cvt_pk_u8_f32","slug":"v_cvt_pk_u8_f32","records":1,"summary":"Convert a single-precision float value from the first input to an unsigned 8-bit integer value and pack the result into one byte of the third input…","page":"https://instructionsets.com/amdgpu/v_cvt_pk_u8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pk_u8_f32.json"},{"mnemonic":"v_cvt_pkaccum_u8_f32","slug":"v_cvt_pkaccum_u8_f32","records":1,"summary":"Convert a single-precision float value in the first input to an unsigned 8-bit integer value and store the result into one byte of the destination…","page":"https://instructionsets.com/amdgpu/v_cvt_pkaccum_u8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pkaccum_u8_f32.json"},{"mnemonic":"v_cvt_pknorm_i16_f16","slug":"v_cvt_pknorm_i16_f16","records":1,"summary":"Convert from two half-precision float inputs to a packed signed normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pknorm_i16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pknorm_i16_f16.json","aliases":["v_cvt_pk_norm_i16_f16"]},{"mnemonic":"v_cvt_pknorm_i16_f32","slug":"v_cvt_pknorm_i16_f32","records":1,"summary":"Convert from two single-precision float inputs to a packed signed normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pknorm_i16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pknorm_i16_f32.json","aliases":["v_cvt_pk_norm_i16_f32"]},{"mnemonic":"v_cvt_pknorm_u16_f16","slug":"v_cvt_pknorm_u16_f16","records":1,"summary":"Convert from two half-precision float inputs to a packed unsigned normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pknorm_u16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pknorm_u16_f16.json","aliases":["v_cvt_pk_norm_u16_f16"]},{"mnemonic":"v_cvt_pknorm_u16_f32","slug":"v_cvt_pknorm_u16_f32","records":1,"summary":"Convert from two single-precision float inputs to a packed unsigned normalized short and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_pknorm_u16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pknorm_u16_f32.json","aliases":["v_cvt_pk_norm_u16_f32"]},{"mnemonic":"v_cvt_pkrtz_f16_f32","slug":"v_cvt_pkrtz_f16_f32","records":1,"summary":"Convert two single-precision float inputs to a packed half-precision float value using round toward zero semantics (ignore the current rounding…","page":"https://instructionsets.com/amdgpu/v_cvt_pkrtz_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_pkrtz_f16_f32.json","aliases":["v_cvt_pk_rtz_f16_f32"]},{"mnemonic":"v_cvt_rpi_i32_f32","slug":"v_cvt_rpi_i32_f32","records":1,"summary":"Convert from a single-precision float input to a signed 32-bit integer value using round to nearest integer semantics (ignore the default rounding…","page":"https://instructionsets.com/amdgpu/v_cvt_rpi_i32_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_rpi_i32_f32.json","aliases":["v_cvt_nearest_i32_f32"]},{"mnemonic":"v_cvt_scale_pk16_bf16_bf6","slug":"v_cvt_scale_pk16_bf16_bf6","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk16_bf16_bf6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk16_bf16_bf6.json"},{"mnemonic":"v_cvt_scale_pk16_bf16_fp6","slug":"v_cvt_scale_pk16_bf16_fp6","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk16_bf16_fp6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk16_bf16_fp6.json"},{"mnemonic":"v_cvt_scale_pk16_f16_bf6","slug":"v_cvt_scale_pk16_f16_bf6","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk16_f16_bf6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk16_f16_bf6.json"},{"mnemonic":"v_cvt_scale_pk16_f16_fp6","slug":"v_cvt_scale_pk16_f16_fp6","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk16_f16_fp6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk16_f16_fp6.json"},{"mnemonic":"v_cvt_scale_pk16_f32_bf6","slug":"v_cvt_scale_pk16_f32_bf6","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk16_f32_bf6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk16_f32_bf6.json"},{"mnemonic":"v_cvt_scale_pk16_f32_fp6","slug":"v_cvt_scale_pk16_f32_fp6","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk16_f32_fp6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk16_f32_fp6.json"},{"mnemonic":"v_cvt_scale_pk8_bf16_bf8","slug":"v_cvt_scale_pk8_bf16_bf8","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk8_bf16_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk8_bf16_bf8.json"},{"mnemonic":"v_cvt_scale_pk8_bf16_fp4","slug":"v_cvt_scale_pk8_bf16_fp4","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk8_bf16_fp4/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk8_bf16_fp4.json"},{"mnemonic":"v_cvt_scale_pk8_bf16_fp8","slug":"v_cvt_scale_pk8_bf16_fp8","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk8_bf16_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk8_bf16_fp8.json"},{"mnemonic":"v_cvt_scale_pk8_f16_bf8","slug":"v_cvt_scale_pk8_f16_bf8","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk8_f16_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk8_f16_bf8.json"},{"mnemonic":"v_cvt_scale_pk8_f16_fp4","slug":"v_cvt_scale_pk8_f16_fp4","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk8_f16_fp4/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk8_f16_fp4.json"},{"mnemonic":"v_cvt_scale_pk8_f16_fp8","slug":"v_cvt_scale_pk8_f16_fp8","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk8_f16_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk8_f16_fp8.json"},{"mnemonic":"v_cvt_scale_pk8_f32_bf8","slug":"v_cvt_scale_pk8_f32_bf8","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk8_f32_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk8_f32_bf8.json"},{"mnemonic":"v_cvt_scale_pk8_f32_fp4","slug":"v_cvt_scale_pk8_f32_fp4","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk8_f32_fp4/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk8_f32_fp4.json"},{"mnemonic":"v_cvt_scale_pk8_f32_fp8","slug":"v_cvt_scale_pk8_f32_fp8","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scale_pk8_f32_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scale_pk8_f32_fp8.json"},{"mnemonic":"v_cvt_scalef32_2xpk16_bf6_f32","slug":"v_cvt_scalef32_2xpk16_bf6_f32","records":1,"summary":"Scale packed 16-component single-precision float vectors from two source inputs using the exponent provided by the third single-precision float…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_2xpk16_bf6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_2xpk16_bf6_f32.json","aliases":["v_cvt_scale_pk_bf6_f32"]},{"mnemonic":"v_cvt_scalef32_2xpk16_fp6_f32","slug":"v_cvt_scalef32_2xpk16_fp6_f32","records":1,"summary":"Scale packed 16-component single-precision float vectors from two source inputs using the exponent provided by the third single-precision float…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_2xpk16_fp6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_2xpk16_fp6_f32.json","aliases":["v_cvt_scale_pk_fp6_f32"]},{"mnemonic":"v_cvt_scalef32_f16_bf8","slug":"v_cvt_scalef32_f16_bf8","records":1,"summary":"Convert from a BF8 float input to a half-precision float value, then scale the value using the exponent provided by the second single-precision float…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_f16_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_f16_bf8.json","aliases":["v_cvt_scale_f16_bf8"]},{"mnemonic":"v_cvt_scalef32_f16_fp8","slug":"v_cvt_scalef32_f16_fp8","records":1,"summary":"Convert from an FP8 float input to a half-precision float value, then scale the value using the exponent provided by the second single-precision…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_f16_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_f16_fp8.json","aliases":["v_cvt_scale_f16_fp8"]},{"mnemonic":"v_cvt_scalef32_f32_bf8","slug":"v_cvt_scalef32_f32_bf8","records":1,"summary":"Convert from a BF8 float input to a single-precision float value, then scale the value using the exponent provided by the second single-precision…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_f32_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_f32_bf8.json","aliases":["v_cvt_scale_f32_bf8"]},{"mnemonic":"v_cvt_scalef32_f32_fp8","slug":"v_cvt_scalef32_f32_fp8","records":1,"summary":"Convert from an FP8 float input to a single-precision float value, then scale the value using the exponent provided by the second single-precision…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_f32_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_f32_fp8.json","aliases":["v_cvt_scale_f32_fp8"]},{"mnemonic":"v_cvt_scalef32_pk16_bf6_bf16","slug":"v_cvt_scalef32_pk16_bf6_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk16_bf6_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk16_bf6_bf16.json"},{"mnemonic":"v_cvt_scalef32_pk16_bf6_f16","slug":"v_cvt_scalef32_pk16_bf6_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk16_bf6_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk16_bf6_f16.json"},{"mnemonic":"v_cvt_scalef32_pk16_bf6_f32","slug":"v_cvt_scalef32_pk16_bf6_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk16_bf6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk16_bf6_f32.json"},{"mnemonic":"v_cvt_scalef32_pk16_fp6_bf16","slug":"v_cvt_scalef32_pk16_fp6_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk16_fp6_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk16_fp6_bf16.json"},{"mnemonic":"v_cvt_scalef32_pk16_fp6_f16","slug":"v_cvt_scalef32_pk16_fp6_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk16_fp6_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk16_fp6_f16.json"},{"mnemonic":"v_cvt_scalef32_pk16_fp6_f32","slug":"v_cvt_scalef32_pk16_fp6_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk16_fp6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk16_fp6_f32.json"},{"mnemonic":"v_cvt_scalef32_pk32_bf16_bf6","slug":"v_cvt_scalef32_pk32_bf16_bf6","records":1,"summary":"Convert from a packed 32-component BF6 float input to a packed BF16 float value, then scale the packed values using the exponent provided by the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_bf16_bf6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_bf16_bf6.json","aliases":["v_cvt_scale_pk_bf16_bf6"]},{"mnemonic":"v_cvt_scalef32_pk32_bf16_fp6","slug":"v_cvt_scalef32_pk32_bf16_fp6","records":1,"summary":"Convert from a packed 32-component FP6 float input to a packed BF16 float value, then scale the packed values using the exponent provided by the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_bf16_fp6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_bf16_fp6.json","aliases":["v_cvt_scale_pk_bf16_fp6"]},{"mnemonic":"v_cvt_scalef32_pk32_bf6_bf16","slug":"v_cvt_scalef32_pk32_bf6_bf16","records":1,"summary":"Scale a packed 32-component BF16 float input using the exponent provided by the second single-precision float input, then convert the values to a…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_bf6_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_bf6_bf16.json","aliases":["v_cvt_scale_pk_bf6_bf16"]},{"mnemonic":"v_cvt_scalef32_pk32_bf6_f16","slug":"v_cvt_scalef32_pk32_bf6_f16","records":1,"summary":"Scale a packed 32-component half-precision float input using the exponent provided by the second single-precision float input, then convert the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_bf6_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_bf6_f16.json","aliases":["v_cvt_scale_pk_bf6_f16"]},{"mnemonic":"v_cvt_scalef32_pk32_bf6_f32","slug":"v_cvt_scalef32_pk32_bf6_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_bf6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_bf6_f32.json"},{"mnemonic":"v_cvt_scalef32_pk32_f16_bf6","slug":"v_cvt_scalef32_pk32_f16_bf6","records":1,"summary":"Convert from a packed 32-component BF6 float input to a packed half-precision float value, then scale the packed values using the exponent provided…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_f16_bf6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_f16_bf6.json","aliases":["v_cvt_scale_pk_f16_bf6"]},{"mnemonic":"v_cvt_scalef32_pk32_f16_fp6","slug":"v_cvt_scalef32_pk32_f16_fp6","records":1,"summary":"Convert from a packed 32-component FP6 float input to a packed half-precision float value, then scale the packed values using the exponent provided…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_f16_fp6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_f16_fp6.json","aliases":["v_cvt_scale_pk_f16_fp6"]},{"mnemonic":"v_cvt_scalef32_pk32_f32_bf6","slug":"v_cvt_scalef32_pk32_f32_bf6","records":1,"summary":"Convert from a packed 32-component BF6 float input to a packed single-precision float value, then scale the packed values using the exponent provided…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_f32_bf6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_f32_bf6.json","aliases":["v_cvt_scale_pk_f32_bf6"]},{"mnemonic":"v_cvt_scalef32_pk32_f32_fp6","slug":"v_cvt_scalef32_pk32_f32_fp6","records":1,"summary":"Convert from a packed 32-component FP6 float input to a packed single-precision float value, then scale the packed values using the exponent provided…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_f32_fp6/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_f32_fp6.json","aliases":["v_cvt_scale_pk_f32_fp6"]},{"mnemonic":"v_cvt_scalef32_pk32_fp6_bf16","slug":"v_cvt_scalef32_pk32_fp6_bf16","records":1,"summary":"Scale a packed 32-component BF16 float input using the exponent provided by the second single-precision float input, then convert the values to a…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_fp6_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_fp6_bf16.json","aliases":["v_cvt_scale_pk_fp6_bf16"]},{"mnemonic":"v_cvt_scalef32_pk32_fp6_f16","slug":"v_cvt_scalef32_pk32_fp6_f16","records":1,"summary":"Scale a packed 32-component half-precision float input using the exponent provided by the second single-precision float input, then convert the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_fp6_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_fp6_f16.json","aliases":["v_cvt_scale_pk_fp6_f16"]},{"mnemonic":"v_cvt_scalef32_pk32_fp6_f32","slug":"v_cvt_scalef32_pk32_fp6_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk32_fp6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk32_fp6_f32.json"},{"mnemonic":"v_cvt_scalef32_pk8_bf8_bf16","slug":"v_cvt_scalef32_pk8_bf8_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk8_bf8_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk8_bf8_bf16.json"},{"mnemonic":"v_cvt_scalef32_pk8_bf8_f16","slug":"v_cvt_scalef32_pk8_bf8_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk8_bf8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk8_bf8_f16.json"},{"mnemonic":"v_cvt_scalef32_pk8_bf8_f32","slug":"v_cvt_scalef32_pk8_bf8_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk8_bf8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk8_bf8_f32.json"},{"mnemonic":"v_cvt_scalef32_pk8_fp4_bf16","slug":"v_cvt_scalef32_pk8_fp4_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk8_fp4_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk8_fp4_bf16.json"},{"mnemonic":"v_cvt_scalef32_pk8_fp4_f16","slug":"v_cvt_scalef32_pk8_fp4_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk8_fp4_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk8_fp4_f16.json"},{"mnemonic":"v_cvt_scalef32_pk8_fp4_f32","slug":"v_cvt_scalef32_pk8_fp4_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk8_fp4_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk8_fp4_f32.json"},{"mnemonic":"v_cvt_scalef32_pk8_fp8_bf16","slug":"v_cvt_scalef32_pk8_fp8_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk8_fp8_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk8_fp8_bf16.json"},{"mnemonic":"v_cvt_scalef32_pk8_fp8_f16","slug":"v_cvt_scalef32_pk8_fp8_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk8_fp8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk8_fp8_f16.json"},{"mnemonic":"v_cvt_scalef32_pk8_fp8_f32","slug":"v_cvt_scalef32_pk8_fp8_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk8_fp8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk8_fp8_f32.json"},{"mnemonic":"v_cvt_scalef32_pk_bf16_bf8","slug":"v_cvt_scalef32_pk_bf16_bf8","records":1,"summary":"Convert from a packed 2-component BF8 float input to a packed BF16 float value, then scale the packed values using the exponent provided by the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_bf16_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_bf16_bf8.json","aliases":["v_cvt_scale_pk_bf16_bf8"]},{"mnemonic":"v_cvt_scalef32_pk_bf16_fp4","slug":"v_cvt_scalef32_pk_bf16_fp4","records":1,"summary":"Convert from a packed 2-component FP4 float input to a packed BF16 float value, then scale the packed values using the exponent provided by the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_bf16_fp4/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_bf16_fp4.json","aliases":["v_cvt_scale_pk_bf16_fp4"]},{"mnemonic":"v_cvt_scalef32_pk_bf16_fp8","slug":"v_cvt_scalef32_pk_bf16_fp8","records":1,"summary":"Convert from a packed 2-component FP8 float input to a packed BF16 float value, then scale the packed values using the exponent provided by the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_bf16_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_bf16_fp8.json","aliases":["v_cvt_scale_pk_bf16_fp8"]},{"mnemonic":"v_cvt_scalef32_pk_bf8_bf16","slug":"v_cvt_scalef32_pk_bf8_bf16","records":1,"summary":"Scale a packed 2-component BF16 float input using the exponent provided by the second single-precision float input, then convert the values to a…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_bf8_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_bf8_bf16.json","aliases":["v_cvt_scale_pk_bf8_bf16"]},{"mnemonic":"v_cvt_scalef32_pk_bf8_f16","slug":"v_cvt_scalef32_pk_bf8_f16","records":1,"summary":"Scale a packed 2-component half-precision float input using the exponent provided by the second single-precision float input, then convert the values…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_bf8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_bf8_f16.json","aliases":["v_cvt_scale_pk_bf8_f16"]},{"mnemonic":"v_cvt_scalef32_pk_bf8_f32","slug":"v_cvt_scalef32_pk_bf8_f32","records":1,"summary":"Scale two single-precision float inputs using the exponent provided by the third single-precision float input, then convert the values to a packed…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_bf8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_bf8_f32.json","aliases":["v_cvt_scale_pk_bf8_f32"]},{"mnemonic":"v_cvt_scalef32_pk_f16_bf8","slug":"v_cvt_scalef32_pk_f16_bf8","records":1,"summary":"Convert from a packed 2-component BF8 float input to a packed half-precision float value, then scale the packed values using the exponent provided by…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_f16_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_f16_bf8.json","aliases":["v_cvt_scale_pk_f16_bf8"]},{"mnemonic":"v_cvt_scalef32_pk_f16_fp4","slug":"v_cvt_scalef32_pk_f16_fp4","records":1,"summary":"Convert from a packed 2-component FP4 float input to a packed half-precision float value, then scale the packed values using the exponent provided by…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_f16_fp4/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_f16_fp4.json","aliases":["v_cvt_scale_pk_f16_fp4"]},{"mnemonic":"v_cvt_scalef32_pk_f16_fp8","slug":"v_cvt_scalef32_pk_f16_fp8","records":1,"summary":"Convert from a packed 2-component FP8 float input to a packed half-precision float value, then scale the packed values using the exponent provided by…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_f16_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_f16_fp8.json","aliases":["v_cvt_scale_pk_f16_fp8"]},{"mnemonic":"v_cvt_scalef32_pk_f32_bf8","slug":"v_cvt_scalef32_pk_f32_bf8","records":1,"summary":"Convert from a packed 2-component BF8 float input to a packed single-precision float value, then scale the packed values using the exponent provided…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_f32_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_f32_bf8.json","aliases":["v_cvt_scale_pk_f32_bf8"]},{"mnemonic":"v_cvt_scalef32_pk_f32_fp4","slug":"v_cvt_scalef32_pk_f32_fp4","records":1,"summary":"Convert from a packed 2-component FP4 float input to a packed single-precision float value, then scale the packed values using the exponent provided…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_f32_fp4/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_f32_fp4.json","aliases":["v_cvt_scale_pk_f32_fp4"]},{"mnemonic":"v_cvt_scalef32_pk_f32_fp8","slug":"v_cvt_scalef32_pk_f32_fp8","records":1,"summary":"Convert from a packed 2-component FP8 float input to a packed single-precision float value, then scale the packed values using the exponent provided…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_f32_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_f32_fp8.json","aliases":["v_cvt_scale_pk_f32_fp8"]},{"mnemonic":"v_cvt_scalef32_pk_fp4_bf16","slug":"v_cvt_scalef32_pk_fp4_bf16","records":1,"summary":"Scale a packed 2-component BF16 float input using the exponent provided by the second single-precision float input, then convert the values to a…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_fp4_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_fp4_bf16.json","aliases":["v_cvt_scale_pk_fp4_bf16"]},{"mnemonic":"v_cvt_scalef32_pk_fp4_f16","slug":"v_cvt_scalef32_pk_fp4_f16","records":1,"summary":"Scale a packed 2-component half-precision float input using the exponent provided by the second single-precision float input, then convert the values…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_fp4_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_fp4_f16.json","aliases":["v_cvt_scale_pk_fp4_f16"]},{"mnemonic":"v_cvt_scalef32_pk_fp4_f32","slug":"v_cvt_scalef32_pk_fp4_f32","records":1,"summary":"Scale two single-precision float inputs using the exponent provided by the third single-precision float input, then convert the values to a packed…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_fp4_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_fp4_f32.json","aliases":["v_cvt_scale_pk_fp4_f32"]},{"mnemonic":"v_cvt_scalef32_pk_fp8_bf16","slug":"v_cvt_scalef32_pk_fp8_bf16","records":1,"summary":"Scale a packed 2-component BF16 float input using the exponent provided by the second single-precision float input, then convert the values to a…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_fp8_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_fp8_bf16.json","aliases":["v_cvt_scale_pk_fp8_bf16"]},{"mnemonic":"v_cvt_scalef32_pk_fp8_f16","slug":"v_cvt_scalef32_pk_fp8_f16","records":1,"summary":"Scale a packed 2-component half-precision float input using the exponent provided by the second single-precision float input, then convert the values…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_fp8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_fp8_f16.json","aliases":["v_cvt_scale_pk_fp8_f16"]},{"mnemonic":"v_cvt_scalef32_pk_fp8_f32","slug":"v_cvt_scalef32_pk_fp8_f32","records":1,"summary":"Scale two single-precision float inputs using the exponent provided by the third single-precision float input, then convert the values to a packed…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_pk_fp8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_pk_fp8_f32.json","aliases":["v_cvt_scale_pk_fp8_f32"]},{"mnemonic":"v_cvt_scalef32_sr_bf8_bf16","slug":"v_cvt_scalef32_sr_bf8_bf16","records":1,"summary":"Scale a BF16 float input using the exponent provided by the third single-precision float input, then convert the values to a BF8 float value with…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_bf8_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_bf8_bf16.json","aliases":["v_cvt_scale_sr_bf8_bf16"]},{"mnemonic":"v_cvt_scalef32_sr_bf8_f16","slug":"v_cvt_scalef32_sr_bf8_f16","records":1,"summary":"Scale a half-precision float input using the exponent provided by the third single-precision float input, then convert the values to a BF8 float…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_bf8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_bf8_f16.json","aliases":["v_cvt_scale_sr_bf8_f16"]},{"mnemonic":"v_cvt_scalef32_sr_bf8_f32","slug":"v_cvt_scalef32_sr_bf8_f32","records":1,"summary":"Scale a single-precision float input using the exponent provided by the third single-precision float input, then convert the values to a BF8 float…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_bf8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_bf8_f32.json","aliases":["v_cvt_scale_sr_bf8_f32"]},{"mnemonic":"v_cvt_scalef32_sr_fp8_bf16","slug":"v_cvt_scalef32_sr_fp8_bf16","records":1,"summary":"Scale a BF16 float input using the exponent provided by the third single-precision float input, then convert the values to an FP8 float value with…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_fp8_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_fp8_bf16.json","aliases":["v_cvt_scale_sr_fp8_bf16"]},{"mnemonic":"v_cvt_scalef32_sr_fp8_f16","slug":"v_cvt_scalef32_sr_fp8_f16","records":1,"summary":"Scale a half-precision float input using the exponent provided by the third single-precision float input, then convert the values to an FP8 float…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_fp8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_fp8_f16.json","aliases":["v_cvt_scale_sr_fp8_f16"]},{"mnemonic":"v_cvt_scalef32_sr_fp8_f32","slug":"v_cvt_scalef32_sr_fp8_f32","records":1,"summary":"Scale a single-precision float input using the exponent provided by the third single-precision float input, then convert the values to an FP8 float…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_fp8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_fp8_f32.json","aliases":["v_cvt_scale_sr_fp8_f32"]},{"mnemonic":"v_cvt_scalef32_sr_pk16_bf6_bf16","slug":"v_cvt_scalef32_sr_pk16_bf6_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk16_bf6_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk16_bf6_bf16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk16_bf6_f16","slug":"v_cvt_scalef32_sr_pk16_bf6_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk16_bf6_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk16_bf6_f16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk16_bf6_f32","slug":"v_cvt_scalef32_sr_pk16_bf6_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk16_bf6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk16_bf6_f32.json"},{"mnemonic":"v_cvt_scalef32_sr_pk16_fp6_bf16","slug":"v_cvt_scalef32_sr_pk16_fp6_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk16_fp6_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk16_fp6_bf16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk16_fp6_f16","slug":"v_cvt_scalef32_sr_pk16_fp6_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk16_fp6_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk16_fp6_f16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk16_fp6_f32","slug":"v_cvt_scalef32_sr_pk16_fp6_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk16_fp6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk16_fp6_f32.json"},{"mnemonic":"v_cvt_scalef32_sr_pk32_bf6_bf16","slug":"v_cvt_scalef32_sr_pk32_bf6_bf16","records":1,"summary":"Scale a packed 32-component BF16 float input using the exponent provided by the third single-precision float input, then convert the values to a…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk32_bf6_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk32_bf6_bf16.json","aliases":["v_cvt_scale_sr_pk_bf6_bf16"]},{"mnemonic":"v_cvt_scalef32_sr_pk32_bf6_f16","slug":"v_cvt_scalef32_sr_pk32_bf6_f16","records":1,"summary":"Scale a packed 32-component half-precision float input using the exponent provided by the third single-precision float input, then convert the values…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk32_bf6_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk32_bf6_f16.json","aliases":["v_cvt_scale_sr_pk_bf6_f16"]},{"mnemonic":"v_cvt_scalef32_sr_pk32_bf6_f32","slug":"v_cvt_scalef32_sr_pk32_bf6_f32","records":1,"summary":"Scale a packed 32-component single-precision float input using the exponent provided by the third single-precision float input, then convert the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk32_bf6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk32_bf6_f32.json","aliases":["v_cvt_scale_sr_pk_bf6_f32"]},{"mnemonic":"v_cvt_scalef32_sr_pk32_fp6_bf16","slug":"v_cvt_scalef32_sr_pk32_fp6_bf16","records":1,"summary":"Scale a packed 32-component BF16 float input using the exponent provided by the third single-precision float input, then convert the values to a…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk32_fp6_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk32_fp6_bf16.json","aliases":["v_cvt_scale_sr_pk_fp6_bf16"]},{"mnemonic":"v_cvt_scalef32_sr_pk32_fp6_f16","slug":"v_cvt_scalef32_sr_pk32_fp6_f16","records":1,"summary":"Scale a packed 32-component half-precision float input using the exponent provided by the third single-precision float input, then convert the values…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk32_fp6_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk32_fp6_f16.json","aliases":["v_cvt_scale_sr_pk_fp6_f16"]},{"mnemonic":"v_cvt_scalef32_sr_pk32_fp6_f32","slug":"v_cvt_scalef32_sr_pk32_fp6_f32","records":1,"summary":"Scale a packed 32-component single-precision float input using the exponent provided by the third single-precision float input, then convert the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk32_fp6_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk32_fp6_f32.json","aliases":["v_cvt_scale_sr_pk_fp6_f32"]},{"mnemonic":"v_cvt_scalef32_sr_pk8_bf8_bf16","slug":"v_cvt_scalef32_sr_pk8_bf8_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk8_bf8_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk8_bf8_bf16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk8_bf8_f16","slug":"v_cvt_scalef32_sr_pk8_bf8_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk8_bf8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk8_bf8_f16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk8_bf8_f32","slug":"v_cvt_scalef32_sr_pk8_bf8_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk8_bf8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk8_bf8_f32.json"},{"mnemonic":"v_cvt_scalef32_sr_pk8_fp4_bf16","slug":"v_cvt_scalef32_sr_pk8_fp4_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk8_fp4_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk8_fp4_bf16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk8_fp4_f16","slug":"v_cvt_scalef32_sr_pk8_fp4_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk8_fp4_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk8_fp4_f16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk8_fp4_f32","slug":"v_cvt_scalef32_sr_pk8_fp4_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk8_fp4_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk8_fp4_f32.json"},{"mnemonic":"v_cvt_scalef32_sr_pk8_fp8_bf16","slug":"v_cvt_scalef32_sr_pk8_fp8_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk8_fp8_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk8_fp8_bf16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk8_fp8_f16","slug":"v_cvt_scalef32_sr_pk8_fp8_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk8_fp8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk8_fp8_f16.json"},{"mnemonic":"v_cvt_scalef32_sr_pk8_fp8_f32","slug":"v_cvt_scalef32_sr_pk8_fp8_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk8_fp8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk8_fp8_f32.json"},{"mnemonic":"v_cvt_scalef32_sr_pk_fp4_bf16","slug":"v_cvt_scalef32_sr_pk_fp4_bf16","records":1,"summary":"Scale a packed 2-component BF16 float input using the exponent provided by the third single-precision float input, then convert the values to a…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk_fp4_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk_fp4_bf16.json","aliases":["v_cvt_scale_sr_pk_fp4_bf16"]},{"mnemonic":"v_cvt_scalef32_sr_pk_fp4_f16","slug":"v_cvt_scalef32_sr_pk_fp4_f16","records":1,"summary":"Scale a packed 2-component half-precision float input using the exponent provided by the third single-precision float input, then convert the values…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk_fp4_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk_fp4_f16.json","aliases":["v_cvt_scale_sr_pk_fp4_f16"]},{"mnemonic":"v_cvt_scalef32_sr_pk_fp4_f32","slug":"v_cvt_scalef32_sr_pk_fp4_f32","records":1,"summary":"Scale a packed 2-component single-precision float input using the exponent provided by the third single-precision float input, then convert the…","page":"https://instructionsets.com/amdgpu/v_cvt_scalef32_sr_pk_fp4_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_scalef32_sr_pk_fp4_f32.json","aliases":["v_cvt_scale_sr_pk_fp4_f32"]},{"mnemonic":"v_cvt_sr_bf16_f32","slug":"v_cvt_sr_bf16_f32","records":1,"summary":"Convert from a single-precision float input to a BF16 value with stochastic rounding using seed data from the second input.","page":"https://instructionsets.com/amdgpu/v_cvt_sr_bf16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_bf16_f32.json"},{"mnemonic":"v_cvt_sr_bf8_f16","slug":"v_cvt_sr_bf8_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_sr_bf8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_bf8_f16.json"},{"mnemonic":"v_cvt_sr_bf8_f32","slug":"v_cvt_sr_bf8_f32","records":1,"summary":"Convert from a single-precision float input to a BF8 value with stochastic rounding using seed data from the second input.","page":"https://instructionsets.com/amdgpu/v_cvt_sr_bf8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_bf8_f32.json"},{"mnemonic":"v_cvt_sr_bf8_f32_gfx12","slug":"v_cvt_sr_bf8_f32_gfx12","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_sr_bf8_f32_gfx12/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_bf8_f32_gfx12.json"},{"mnemonic":"v_cvt_sr_f16_f32","slug":"v_cvt_sr_f16_f32","records":1,"summary":"Convert from a single-precision float input to a half-precision value with stochastic rounding using seed data from the second input.","page":"https://instructionsets.com/amdgpu/v_cvt_sr_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_f16_f32.json"},{"mnemonic":"v_cvt_sr_fp8_f16","slug":"v_cvt_sr_fp8_f16","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_sr_fp8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_fp8_f16.json"},{"mnemonic":"v_cvt_sr_fp8_f32","slug":"v_cvt_sr_fp8_f32","records":1,"summary":"Convert from a single-precision float input to an FP8 value with stochastic rounding using seed data from the second input.","page":"https://instructionsets.com/amdgpu/v_cvt_sr_fp8_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_fp8_f32.json"},{"mnemonic":"v_cvt_sr_fp8_f32_gfx12","slug":"v_cvt_sr_fp8_f32_gfx12","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_sr_fp8_f32_gfx12/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_fp8_f32_gfx12.json"},{"mnemonic":"v_cvt_sr_fp8_f32_gfx1250","slug":"v_cvt_sr_fp8_f32_gfx1250","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_sr_fp8_f32_gfx1250/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_fp8_f32_gfx1250.json"},{"mnemonic":"v_cvt_sr_pk_bf16_f32","slug":"v_cvt_sr_pk_bf16_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_sr_pk_bf16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_pk_bf16_f32.json"},{"mnemonic":"v_cvt_sr_pk_f16_f32","slug":"v_cvt_sr_pk_f16_f32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16/f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_cvt_sr_pk_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_sr_pk_f16_f32.json"},{"mnemonic":"v_cvt_u16_f16","slug":"v_cvt_u16_f16","records":1,"summary":"Convert from a half-precision float input to an unsigned 16-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_u16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_u16_f16.json"},{"mnemonic":"v_cvt_u32_f32","slug":"v_cvt_u32_f32","records":1,"summary":"Convert from a single-precision float input to an unsigned 32-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_u32_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_u32_f32.json"},{"mnemonic":"v_cvt_u32_f64","slug":"v_cvt_u32_f64","records":1,"summary":"Convert from a double-precision float input to an unsigned 32-bit integer value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_u32_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_u32_f64.json"},{"mnemonic":"v_cvt_u32_u16","slug":"v_cvt_u32_u16","records":1,"summary":"Convert from an unsigned 16-bit integer input to an unsigned 32-bit integer value using zero extension and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_cvt_u32_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_cvt_u32_u16.json"},{"mnemonic":"v_div_fixup_f16","slug":"v_div_fixup_f16","records":1,"summary":"Given a half-precision float quotient in the first input, a denominator in the second input and a numerator in the third input, detect and apply…","page":"https://instructionsets.com/amdgpu/v_div_fixup_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_div_fixup_f16.json"},{"mnemonic":"v_div_fixup_f16_gfx9","slug":"v_div_fixup_f16_gfx9","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_div_fixup_f16_gfx9/","api":"https://instructionsets.com/api/v1/amdgpu/v_div_fixup_f16_gfx9.json"},{"mnemonic":"v_div_fixup_f32","slug":"v_div_fixup_f32","records":1,"summary":"Given a single-precision float quotient in the first input, a denominator in the second input and a numerator in the third input, detect and apply…","page":"https://instructionsets.com/amdgpu/v_div_fixup_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_div_fixup_f32.json"},{"mnemonic":"v_div_fixup_f64","slug":"v_div_fixup_f64","records":1,"summary":"Given a double-precision float quotient in the first input, a denominator in the second input and a numerator in the third input, detect and apply…","page":"https://instructionsets.com/amdgpu/v_div_fixup_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_div_fixup_f64.json"},{"mnemonic":"v_div_fixup_legacy_f16","slug":"v_div_fixup_legacy_f16","records":1,"summary":"Half precision division fixup. Has non-standard rule for OPSEL.","page":"https://instructionsets.com/amdgpu/v_div_fixup_legacy_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_div_fixup_legacy_f16.json"},{"mnemonic":"v_div_fmas_f32","slug":"v_div_fmas_f32","records":1,"summary":"Multiply two single-precision float inputs and add a third input using fused multiply add, then scale the exponent of the result by a fixed factor if…","page":"https://instructionsets.com/amdgpu/v_div_fmas_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_div_fmas_f32.json"},{"mnemonic":"v_div_fmas_f64","slug":"v_div_fmas_f64","records":1,"summary":"Multiply two double-precision float inputs and add a third input using fused multiply add, then scale the exponent of the result by a fixed factor if…","page":"https://instructionsets.com/amdgpu/v_div_fmas_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_div_fmas_f64.json"},{"mnemonic":"v_div_scale_f32","slug":"v_div_scale_f32","records":1,"summary":"Given a single-precision float value to scale in the first input, a denominator in the second input and a numerator in the third input, scale the…","page":"https://instructionsets.com/amdgpu/v_div_scale_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_div_scale_f32.json"},{"mnemonic":"v_div_scale_f64","slug":"v_div_scale_f64","records":1,"summary":"Given a double-precision float value to scale in the first input, a denominator in the second input and a numerator in the third input, scale the…","page":"https://instructionsets.com/amdgpu/v_div_scale_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_div_scale_f64.json"},{"mnemonic":"v_dot2_bf16_bf16","slug":"v_dot2_bf16_bf16","records":1,"summary":"Compute the dot product of two packed 2-D BF16 float inputs, add the third input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dot2_bf16_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot2_bf16_bf16.json"},{"mnemonic":"v_dot2_f16_f16","slug":"v_dot2_f16_f16","records":1,"summary":"Compute the dot product of two packed 2-D half-precision float inputs, add the third input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dot2_f16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot2_f16_f16.json"},{"mnemonic":"v_dot2_f32_bf16","slug":"v_dot2_f32_bf16","records":1,"summary":"Calculate the dot product of BF16 float 2-vectors from the first and second inputs, convert the product to single-precision float format, add the…","page":"https://instructionsets.com/amdgpu/v_dot2_f32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot2_f32_bf16.json"},{"mnemonic":"v_dot2_f32_f16","slug":"v_dot2_f32_f16","records":1,"summary":"Compute the dot product of two packed 2-D half-precision float inputs in the single-precision float domain, add a single-precision float value from…","page":"https://instructionsets.com/amdgpu/v_dot2_f32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot2_f32_f16.json"},{"mnemonic":"v_dot2_i32_i16","slug":"v_dot2_i32_i16","records":1,"summary":"Compute the dot product of two packed 2-D signed 16-bit integer inputs in the signed 32-bit integer domain, add a signed 32-bit integer value from…","page":"https://instructionsets.com/amdgpu/v_dot2_i32_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot2_i32_i16.json"},{"mnemonic":"v_dot2_u32_u16","slug":"v_dot2_u32_u16","records":1,"summary":"Compute the dot product of two packed 2-D unsigned 16-bit integer inputs in the unsigned 32-bit integer domain, add an unsigned 32-bit integer value…","page":"https://instructionsets.com/amdgpu/v_dot2_u32_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot2_u32_u16.json"},{"mnemonic":"v_dot2c_f32_bf16","slug":"v_dot2c_f32_bf16","records":1,"summary":"Compute the dot product of two packed 2-D BF16 float inputs in the single-precision float domain and accumulate with the single-precision float value…","page":"https://instructionsets.com/amdgpu/v_dot2c_f32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot2c_f32_bf16.json"},{"mnemonic":"v_dot2c_f32_f16","slug":"v_dot2c_f32_f16","records":1,"summary":"Compute the dot product of two packed 2-D half-precision float inputs in the single-precision float domain and accumulate with the single-precision…","page":"https://instructionsets.com/amdgpu/v_dot2c_f32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot2c_f32_f16.json","aliases":["v_dot2acc_f32_f16"]},{"mnemonic":"v_dot2c_i32_i16","slug":"v_dot2c_i32_i16","records":1,"summary":"Compute the dot product of two packed 2-D signed 16-bit integer inputs in the signed 32-bit integer domain and accumulate with the signed 32-bit…","page":"https://instructionsets.com/amdgpu/v_dot2c_i32_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot2c_i32_i16.json"},{"mnemonic":"v_dot4_f32_bf8_bf8","slug":"v_dot4_f32_bf8_bf8","records":1,"summary":"Compute the dot product of two packed 4-D BF8 float inputs in the single-precision float domain, add a single-precision float value from the third…","page":"https://instructionsets.com/amdgpu/v_dot4_f32_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot4_f32_bf8_bf8.json"},{"mnemonic":"v_dot4_f32_bf8_fp8","slug":"v_dot4_f32_bf8_fp8","records":1,"summary":"Compute the dot product of a packed 4-D BF8 float input and a packed 4-D FP8 float input in the single-precision float domain, add a single-precision…","page":"https://instructionsets.com/amdgpu/v_dot4_f32_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot4_f32_bf8_fp8.json"},{"mnemonic":"v_dot4_f32_fp8_bf8","slug":"v_dot4_f32_fp8_bf8","records":1,"summary":"Compute the dot product of a packed 4-D FP8 float input and a packed 4-D BF8 float input in the single-precision float domain, add a single-precision…","page":"https://instructionsets.com/amdgpu/v_dot4_f32_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot4_f32_fp8_bf8.json"},{"mnemonic":"v_dot4_f32_fp8_fp8","slug":"v_dot4_f32_fp8_fp8","records":1,"summary":"Compute the dot product of two packed 4-D FP8 float inputs in the single-precision float domain, add a single-precision float value from the third…","page":"https://instructionsets.com/amdgpu/v_dot4_f32_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot4_f32_fp8_fp8.json"},{"mnemonic":"v_dot4_i32_i8","slug":"v_dot4_i32_i8","records":1,"summary":"Compute the dot product of two packed 4-D signed 8-bit integer inputs in the signed 32-bit integer domain, add a signed 32-bit integer value from the…","page":"https://instructionsets.com/amdgpu/v_dot4_i32_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot4_i32_i8.json","aliases":["v_dot4_i32_iu8"]},{"mnemonic":"v_dot4_i32_iu8","slug":"v_dot4_i32_iu8","records":1,"summary":"Compute the dot product of two packed 4-D signed or unsigned 8-bit integer inputs in the signed 32-bit integer domain, add a signed 32-bit integer…","page":"https://instructionsets.com/amdgpu/v_dot4_i32_iu8/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot4_i32_iu8.json","aliases":["v_dot4_i32_i8"]},{"mnemonic":"v_dot4_u32_u8","slug":"v_dot4_u32_u8","records":1,"summary":"Compute the dot product of two packed 4-D unsigned 8-bit integer inputs in the unsigned 32-bit integer domain, add an unsigned 32-bit integer value…","page":"https://instructionsets.com/amdgpu/v_dot4_u32_u8/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot4_u32_u8.json"},{"mnemonic":"v_dot4c_i32_i8","slug":"v_dot4c_i32_i8","records":1,"summary":"Compute the dot product of two packed 4-D signed 8-bit integer inputs in the signed 32-bit integer domain and accumulate with the signed 32-bit…","page":"https://instructionsets.com/amdgpu/v_dot4c_i32_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot4c_i32_i8.json"},{"mnemonic":"v_dot8_i32_i4","slug":"v_dot8_i32_i4","records":1,"summary":"Compute the dot product of two packed 8-D signed 4-bit integer inputs in the signed 32-bit integer domain, add a signed 32-bit integer value from the…","page":"https://instructionsets.com/amdgpu/v_dot8_i32_i4/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot8_i32_i4.json","aliases":["v_dot8_i32_iu4"]},{"mnemonic":"v_dot8_i32_iu4","slug":"v_dot8_i32_iu4","records":1,"summary":"Compute the dot product of two packed 8-D signed or unsigned 4-bit integer inputs in the signed 32-bit integer domain, add a signed 32-bit integer…","page":"https://instructionsets.com/amdgpu/v_dot8_i32_iu4/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot8_i32_iu4.json","aliases":["v_dot8_i32_i4"]},{"mnemonic":"v_dot8_u32_u4","slug":"v_dot8_u32_u4","records":1,"summary":"Compute the dot product of two packed 8-D unsigned 4-bit integer inputs in the unsigned 32-bit integer domain, add an unsigned 32-bit integer value…","page":"https://instructionsets.com/amdgpu/v_dot8_u32_u4/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot8_u32_u4.json"},{"mnemonic":"v_dot8c_i32_i4","slug":"v_dot8c_i32_i4","records":1,"summary":"Compute the dot product of two packed 8-D signed 4-bit integer inputs in the signed 32-bit integer domain and accumulate with the signed 32-bit…","page":"https://instructionsets.com/amdgpu/v_dot8c_i32_i4/","api":"https://instructionsets.com/api/v1/amdgpu/v_dot8c_i32_i4.json"},{"mnemonic":"v_dual_add_f32","slug":"v_dual_add_f32","records":1,"summary":"Add two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_add_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_add_f32.json"},{"mnemonic":"v_dual_add_f64","slug":"v_dual_add_f64","records":1,"summary":"Add two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_add_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_add_f64.json"},{"mnemonic":"v_dual_add_nc_u32","slug":"v_dual_add_nc_u32","records":1,"summary":"Add two unsigned 32-bit integer inputs and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_dual_add_nc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_add_nc_u32.json"},{"mnemonic":"v_dual_and_b32","slug":"v_dual_and_b32","records":1,"summary":"Calculate bitwise AND on two vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_and_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_and_b32.json"},{"mnemonic":"v_dual_ashrrev_i32","slug":"v_dual_ashrrev_i32","records":1,"summary":"Given a shift count in the first vector input, calculate the arithmetic shift right (preserving sign bit) of the second vector input and store the…","page":"https://instructionsets.com/amdgpu/v_dual_ashrrev_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_ashrrev_i32.json"},{"mnemonic":"v_dual_bitop2_b32","slug":"v_dual_bitop2_b32","records":1,"summary":"Compute a 2-operand generic bitwise operation using a truth table encoded into the instruction.","page":"https://instructionsets.com/amdgpu/v_dual_bitop2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_bitop2_b32.json"},{"mnemonic":"v_dual_cndmask_b32","slug":"v_dual_cndmask_b32","records":1,"summary":"Copy data from one of two inputs based on the per-lane condition code and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_cndmask_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_cndmask_b32.json"},{"mnemonic":"v_dual_dot2acc_f32_bf16","slug":"v_dual_dot2acc_f32_bf16","records":1,"summary":"Dot product of packed brain-float values, accumulate with destination. The initial value in D is used as S2.","page":"https://instructionsets.com/amdgpu/v_dual_dot2acc_f32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_dot2acc_f32_bf16.json"},{"mnemonic":"v_dual_dot2acc_f32_f16","slug":"v_dual_dot2acc_f32_f16","records":1,"summary":"Compute the dot product of two packed 2-D half-precision float inputs in the single-precision float domain and accumulate the resulting…","page":"https://instructionsets.com/amdgpu/v_dual_dot2acc_f32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_dot2acc_f32_f16.json"},{"mnemonic":"v_dual_fma_f32","slug":"v_dual_fma_f32","records":1,"summary":"Multiply two single-precision float inputs and add a third input using fused multiply add, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_fma_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_fma_f32.json"},{"mnemonic":"v_dual_fma_f64","slug":"v_dual_fma_f64","records":1,"summary":"Multiply two double-precision float inputs and add a third input using fused multiply add, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_fma_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_fma_f64.json"},{"mnemonic":"v_dual_fmaak_f32","slug":"v_dual_fmaak_f32","records":1,"summary":"Multiply two single-precision float inputs and add a literal constant using fused multiply add, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_fmaak_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_fmaak_f32.json"},{"mnemonic":"v_dual_fmac_f32","slug":"v_dual_fmac_f32","records":1,"summary":"Multiply two single-precision float inputs and accumulate the result into the destination register using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_dual_fmac_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_fmac_f32.json"},{"mnemonic":"v_dual_fmamk_f32","slug":"v_dual_fmamk_f32","records":1,"summary":"Multiply a single-precision float input with a literal constant and add a second single-precision float input using fused multiply add, and store the…","page":"https://instructionsets.com/amdgpu/v_dual_fmamk_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_fmamk_f32.json"},{"mnemonic":"v_dual_lshlrev_b32","slug":"v_dual_lshlrev_b32","records":1,"summary":"Given a shift count in the first vector input, calculate the logical shift left of the second vector input and store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_dual_lshlrev_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_lshlrev_b32.json"},{"mnemonic":"v_dual_lshrrev_b32","slug":"v_dual_lshrrev_b32","records":1,"summary":"Given a shift count in the first vector input, calculate the logical shift right of the second vector input and store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_dual_lshrrev_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_lshrrev_b32.json"},{"mnemonic":"v_dual_max_f32","slug":"v_dual_max_f32","records":1,"summary":"Select the maximum of two single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_max_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_max_f32.json"},{"mnemonic":"v_dual_max_i32","slug":"v_dual_max_i32","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_max_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_max_i32.json"},{"mnemonic":"v_dual_max_num_f32","slug":"v_dual_max_num_f32","records":1,"summary":"Select the IEEE maximumNumber() of two single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_max_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_max_num_f32.json"},{"mnemonic":"v_dual_max_num_f64","slug":"v_dual_max_num_f64","records":1,"summary":"Select the IEEE maximumNumber() of two double-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_max_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_max_num_f64.json"},{"mnemonic":"v_dual_min_f32","slug":"v_dual_min_f32","records":1,"summary":"Select the minimum of two single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_min_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_min_f32.json"},{"mnemonic":"v_dual_min_i32","slug":"v_dual_min_i32","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_min_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_min_i32.json"},{"mnemonic":"v_dual_min_num_f32","slug":"v_dual_min_num_f32","records":1,"summary":"Select the IEEE minimumNumber() of two single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_min_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_min_num_f32.json"},{"mnemonic":"v_dual_min_num_f64","slug":"v_dual_min_num_f64","records":1,"summary":"Select the IEEE minimumNumber() of two double-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_min_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_min_num_f64.json"},{"mnemonic":"v_dual_mov_b32","slug":"v_dual_mov_b32","records":1,"summary":"Move 32-bit data from a vector input into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_mov_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_mov_b32.json"},{"mnemonic":"v_dual_mul_dx9_zero_f32","slug":"v_dual_mul_dx9_zero_f32","records":1,"summary":"Multiply two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_mul_dx9_zero_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_mul_dx9_zero_f32.json"},{"mnemonic":"v_dual_mul_f32","slug":"v_dual_mul_f32","records":1,"summary":"Multiply two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_mul_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_mul_f32.json"},{"mnemonic":"v_dual_mul_f64","slug":"v_dual_mul_f64","records":1,"summary":"Multiply two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_mul_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_mul_f64.json"},{"mnemonic":"v_dual_sub_f32","slug":"v_dual_sub_f32","records":1,"summary":"Subtract the second floating point input from the first input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_sub_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_sub_f32.json"},{"mnemonic":"v_dual_sub_nc_u32","slug":"v_dual_sub_nc_u32","records":1,"summary":"Subtract the second unsigned 32-bit integer input from the first input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_dual_sub_nc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_sub_nc_u32.json"},{"mnemonic":"v_dual_subrev_f32","slug":"v_dual_subrev_f32","records":1,"summary":"Subtract the first floating point input from the second input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_dual_subrev_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_dual_subrev_f32.json"},{"mnemonic":"v_exp_bf16","slug":"v_exp_bf16","records":1,"summary":"AMDGPU VOP1 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_exp_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_exp_bf16.json"},{"mnemonic":"v_exp_f16","slug":"v_exp_f16","records":1,"summary":"Calculate 2 raised to the power of the half-precision float input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_exp_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_exp_f16.json"},{"mnemonic":"v_exp_f32","slug":"v_exp_f32","records":1,"summary":"Calculate 2 raised to the power of the single-precision float input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_exp_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_exp_f32.json"},{"mnemonic":"v_exp_legacy_f32","slug":"v_exp_legacy_f32","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_exp_legacy_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_exp_legacy_f32.json"},{"mnemonic":"v_ffbh_i32","slug":"v_ffbh_i32","records":1,"summary":"Count the number of leading bits that are the same as the sign bit of a vector input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_ffbh_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_ffbh_i32.json","aliases":["v_cls_i32"]},{"mnemonic":"v_ffbh_u32","slug":"v_ffbh_u32","records":1,"summary":"Count the number of leading \"0\" bits before the first \"1\" in a vector input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_ffbh_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_ffbh_u32.json","aliases":["v_clz_i32_u32"]},{"mnemonic":"v_ffbl_b32","slug":"v_ffbl_b32","records":1,"summary":"Count the number of trailing \"0\" bits before the first \"1\" in a vector input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_ffbl_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_ffbl_b32.json","aliases":["v_ctz_i32_b32"]},{"mnemonic":"v_floor_f16","slug":"v_floor_f16","records":1,"summary":"Round the half-precision float input down to previous integer and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_floor_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_floor_f16.json"},{"mnemonic":"v_floor_f32","slug":"v_floor_f32","records":1,"summary":"Round the single-precision float input down to previous integer and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_floor_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_floor_f32.json"},{"mnemonic":"v_floor_f64","slug":"v_floor_f64","records":1,"summary":"Round the double-precision float input down to previous integer and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_floor_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_floor_f64.json"},{"mnemonic":"v_fma_dx9_zero_f32","slug":"v_fma_dx9_zero_f32","records":1,"summary":"Multiply and add single-precision values. Follows DX9 rules where 0.0 times anything produces 0.0.","page":"https://instructionsets.com/amdgpu/v_fma_dx9_zero_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_dx9_zero_f32.json","aliases":["v_fma_legacy_f32"]},{"mnemonic":"v_fma_f16","slug":"v_fma_f16","records":1,"summary":"Multiply two half-precision float inputs and add a third input using fused multiply add, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_fma_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_f16.json"},{"mnemonic":"v_fma_f16_gfx9","slug":"v_fma_f16_gfx9","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fma_f16_gfx9/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_f16_gfx9.json"},{"mnemonic":"v_fma_f32","slug":"v_fma_f32","records":1,"summary":"Per-lane single-precision fused multiply-add.","page":"https://instructionsets.com/amdgpu/v_fma_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_f32.json"},{"mnemonic":"v_fma_f64","slug":"v_fma_f64","records":1,"summary":"Multiply two double-precision float inputs and add a third input using fused multiply add, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_fma_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_f64.json"},{"mnemonic":"v_fma_legacy_f16","slug":"v_fma_legacy_f16","records":1,"summary":"Fused half precision multiply add. Implements IEEE rules and non-standard rule for OPSEL.","page":"https://instructionsets.com/amdgpu/v_fma_legacy_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_legacy_f16.json"},{"mnemonic":"v_fma_legacy_f32","slug":"v_fma_legacy_f32","records":1,"summary":"Multiply and add single-precision values. Follows DX9 rules where 0.0 times anything produces 0.0.","page":"https://instructionsets.com/amdgpu/v_fma_legacy_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_legacy_f32.json","aliases":["v_fma_dx9_zero_f32"]},{"mnemonic":"v_fma_mix_bf16_t16","slug":"v_fma_mix_bf16_t16","records":1,"summary":"AMDGPU VOP3P vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fma_mix_bf16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_mix_bf16_t16.json"},{"mnemonic":"v_fma_mix_f16_t16","slug":"v_fma_mix_f16_t16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fma_mix_f16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_mix_f16_t16.json"},{"mnemonic":"v_fma_mix_f32","slug":"v_fma_mix_f32","records":1,"summary":"Multiply two inputs and add a third input using fused multiply add where the inputs are a mix of half-precision float and single-precision float…","page":"https://instructionsets.com/amdgpu/v_fma_mix_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_mix_f32.json","aliases":["v_fma_mix_f32_f16"]},{"mnemonic":"v_fma_mix_f32_bf16","slug":"v_fma_mix_f32_bf16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fma_mix_f32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_mix_f32_bf16.json"},{"mnemonic":"v_fma_mixhi_bf16","slug":"v_fma_mixhi_bf16","records":1,"summary":"AMDGPU VOP3P vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fma_mixhi_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_mixhi_bf16.json"},{"mnemonic":"v_fma_mixhi_f16","slug":"v_fma_mixhi_f16","records":1,"summary":"Multiply two inputs and add a third input using fused multiply add where the inputs are a mix of half-precision float and single-precision float…","page":"https://instructionsets.com/amdgpu/v_fma_mixhi_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_mixhi_f16.json"},{"mnemonic":"v_fma_mixlo_bf16","slug":"v_fma_mixlo_bf16","records":1,"summary":"AMDGPU VOP3P vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fma_mixlo_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_mixlo_bf16.json"},{"mnemonic":"v_fma_mixlo_f16","slug":"v_fma_mixlo_f16","records":1,"summary":"Multiply two inputs and add a third input using fused multiply add where the inputs are a mix of half-precision float and single-precision float…","page":"https://instructionsets.com/amdgpu/v_fma_mixlo_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fma_mixlo_f16.json"},{"mnemonic":"v_fmaak_f16","slug":"v_fmaak_f16","records":1,"summary":"Multiply two half-precision float inputs and add a literal constant using fused multiply add, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_fmaak_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmaak_f16.json"},{"mnemonic":"v_fmaak_f16_fake16","slug":"v_fmaak_f16_fake16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fmaak_f16_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmaak_f16_fake16.json"},{"mnemonic":"v_fmaak_f16_t16","slug":"v_fmaak_f16_t16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fmaak_f16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmaak_f16_t16.json"},{"mnemonic":"v_fmaak_f32","slug":"v_fmaak_f32","records":1,"summary":"Multiply two single-precision float inputs and add a literal constant using fused multiply add, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_fmaak_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmaak_f32.json"},{"mnemonic":"v_fmaak_f64","slug":"v_fmaak_f64","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fmaak_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmaak_f64.json"},{"mnemonic":"v_fmac_f16","slug":"v_fmac_f16","records":1,"summary":"Multiply two half-precision float inputs and accumulate the result into the destination register using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_fmac_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmac_f16.json"},{"mnemonic":"v_fmac_f16_fake16","slug":"v_fmac_f16_fake16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fmac_f16_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmac_f16_fake16.json"},{"mnemonic":"v_fmac_f16_t16","slug":"v_fmac_f16_t16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fmac_f16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmac_f16_t16.json"},{"mnemonic":"v_fmac_f32","slug":"v_fmac_f32","records":1,"summary":"Multiply two floating point inputs and accumulate the result into the destination register using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_fmac_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmac_f32.json"},{"mnemonic":"v_fmac_f64","slug":"v_fmac_f64","records":1,"summary":"Multiply two floating point inputs and accumulate the result into the destination register using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_fmac_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmac_f64.json"},{"mnemonic":"v_fmac_legacy_f32","slug":"v_fmac_legacy_f32","records":1,"summary":"Multiply two single-precision values and accumulate the result with the destination. Follows DX9 rules where 0.0 times anything produces 0.0.","page":"https://instructionsets.com/amdgpu/v_fmac_legacy_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmac_legacy_f32.json","aliases":["v_fmac_dx9_zero_f32"]},{"mnemonic":"v_fmamk_f16","slug":"v_fmamk_f16","records":1,"summary":"Multiply a half-precision float input with a literal constant and add a second half-precision float input using fused multiply add, and store the…","page":"https://instructionsets.com/amdgpu/v_fmamk_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmamk_f16.json"},{"mnemonic":"v_fmamk_f16_fake16","slug":"v_fmamk_f16_fake16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fmamk_f16_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmamk_f16_fake16.json"},{"mnemonic":"v_fmamk_f16_t16","slug":"v_fmamk_f16_t16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fmamk_f16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmamk_f16_t16.json"},{"mnemonic":"v_fmamk_f32","slug":"v_fmamk_f32","records":1,"summary":"Multiply a single-precision float input with a literal constant and add a second single-precision float input using fused multiply add, and store the…","page":"https://instructionsets.com/amdgpu/v_fmamk_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmamk_f32.json"},{"mnemonic":"v_fmamk_f64","slug":"v_fmamk_f64","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_fmamk_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_fmamk_f64.json"},{"mnemonic":"v_fract_f16","slug":"v_fract_f16","records":1,"summary":"Compute the fractional portion of a half-precision float input and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_fract_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_fract_f16.json"},{"mnemonic":"v_fract_f32","slug":"v_fract_f32","records":1,"summary":"Compute the fractional portion of a single-precision float input and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_fract_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_fract_f32.json"},{"mnemonic":"v_fract_f64","slug":"v_fract_f64","records":1,"summary":"Compute the fractional portion of a double-precision float input and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_fract_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_fract_f64.json"},{"mnemonic":"v_frexp_exp_i16_f16","slug":"v_frexp_exp_i16_f16","records":1,"summary":"Extract the exponent of a half-precision float input and store the result as a signed 16-bit integer into a vector register.","page":"https://instructionsets.com/amdgpu/v_frexp_exp_i16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_frexp_exp_i16_f16.json"},{"mnemonic":"v_frexp_exp_i32_f32","slug":"v_frexp_exp_i32_f32","records":1,"summary":"Extract the exponent of a single-precision float input and store the result as a signed 32-bit integer into a vector register.","page":"https://instructionsets.com/amdgpu/v_frexp_exp_i32_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_frexp_exp_i32_f32.json"},{"mnemonic":"v_frexp_exp_i32_f64","slug":"v_frexp_exp_i32_f64","records":1,"summary":"Extract the exponent of a double-precision float input and store the result as a signed 32-bit integer into a vector register.","page":"https://instructionsets.com/amdgpu/v_frexp_exp_i32_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_frexp_exp_i32_f64.json"},{"mnemonic":"v_frexp_mant_f16","slug":"v_frexp_mant_f16","records":1,"summary":"Extract the binary significand, or mantissa, of a half-precision float input and store the result as a half- precision float into a vector register.","page":"https://instructionsets.com/amdgpu/v_frexp_mant_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_frexp_mant_f16.json"},{"mnemonic":"v_frexp_mant_f32","slug":"v_frexp_mant_f32","records":1,"summary":"Extract the binary significand, or mantissa, of a single-precision float input and store the result as a single- precision float into a vector…","page":"https://instructionsets.com/amdgpu/v_frexp_mant_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_frexp_mant_f32.json"},{"mnemonic":"v_frexp_mant_f64","slug":"v_frexp_mant_f64","records":1,"summary":"Extract the binary significand, or mantissa, of a double-precision float input and store the result as a double- precision float into a vector…","page":"https://instructionsets.com/amdgpu/v_frexp_mant_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_frexp_mant_f64.json"},{"mnemonic":"v_interp_mov_f32","slug":"v_interp_mov_f32","records":1,"summary":"Given an attribute specifier and a parameter ID (P0, P10 or P20), load one of the parameter values from the local data share into a vector register.","page":"https://instructionsets.com/amdgpu/v_interp_mov_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_mov_f32.json"},{"mnemonic":"v_interp_p10_f16_f32","slug":"v_interp_p10_f16_f32","records":1,"summary":"Given a half-precision float P10 parameter of an attribute, a single-precision float I coordinate and a half-precision float P0 parameter as inputs…","page":"https://instructionsets.com/amdgpu/v_interp_p10_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p10_f16_f32.json"},{"mnemonic":"v_interp_p10_f32","slug":"v_interp_p10_f32","records":1,"summary":"Given the P10 parameter of an attribute, the I coordinate and the P0 parameter as single-precision float inputs, compute the first part of parameter…","page":"https://instructionsets.com/amdgpu/v_interp_p10_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p10_f32.json"},{"mnemonic":"v_interp_p10_rtz_f16_f32","slug":"v_interp_p10_rtz_f16_f32","records":1,"summary":"Given a half-precision float P10 parameter of an attribute, a single-precision float I coordinate and a half-precision float P0 parameter as inputs…","page":"https://instructionsets.com/amdgpu/v_interp_p10_rtz_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p10_rtz_f16_f32.json"},{"mnemonic":"v_interp_p1_f32","slug":"v_interp_p1_f32","records":1,"summary":"Given the I coordinate in a vector register and an attribute specifier, load parameter data from the local data share, compute the first part of…","page":"https://instructionsets.com/amdgpu/v_interp_p1_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p1_f32.json"},{"mnemonic":"v_interp_p1ll_f16","slug":"v_interp_p1ll_f16","records":1,"summary":"Given a single-precision float I coordinate in a vector register and an attribute specifier, load two half-precision float parameter values from the…","page":"https://instructionsets.com/amdgpu/v_interp_p1ll_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p1ll_f16.json"},{"mnemonic":"v_interp_p1lv_f16","slug":"v_interp_p1lv_f16","records":1,"summary":"Given a single-precision float I coordinate in a vector register, a half-precision float P0 value in another vector register, and an attribute…","page":"https://instructionsets.com/amdgpu/v_interp_p1lv_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p1lv_f16.json"},{"mnemonic":"v_interp_p2_f16","slug":"v_interp_p2_f16","records":1,"summary":"Given a single-precision float J coordinate in a vector register, an attribute specifier and the result of a prior V_INTERP_P1_F32 in another vector…","page":"https://instructionsets.com/amdgpu/v_interp_p2_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p2_f16.json"},{"mnemonic":"v_interp_p2_f16_f32","slug":"v_interp_p2_f16_f32","records":1,"summary":"Given a half-precision float P20 parameter of an attribute, a single-precision float J coordinate and the result of a prior V_INTERP_P10_F16_F32…","page":"https://instructionsets.com/amdgpu/v_interp_p2_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p2_f16_f32.json"},{"mnemonic":"v_interp_p2_f16_opsel","slug":"v_interp_p2_f16_opsel","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_interp_p2_f16_opsel/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p2_f16_opsel.json"},{"mnemonic":"v_interp_p2_f32","slug":"v_interp_p2_f32","records":1,"summary":"Given the J coordinate in a vector register, an attribute specifier and the result of a prior V_INTERP_P1_F32 in the destination vector register…","page":"https://instructionsets.com/amdgpu/v_interp_p2_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p2_f32.json","aliases":["v_interp_p2_new_f32"]},{"mnemonic":"v_interp_p2_legacy_f16","slug":"v_interp_p2_legacy_f16","records":1,"summary":"Half-precision interpolation.","page":"https://instructionsets.com/amdgpu/v_interp_p2_legacy_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p2_legacy_f16.json"},{"mnemonic":"v_interp_p2_rtz_f16_f32","slug":"v_interp_p2_rtz_f16_f32","records":1,"summary":"Given a half-precision float P20 parameter of an attribute, a single-precision float J coordinate and the result of a prior V_INTERP_P10_RTZ_F16_F32…","page":"https://instructionsets.com/amdgpu/v_interp_p2_rtz_f16_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_interp_p2_rtz_f16_f32.json"},{"mnemonic":"v_ldexp_f16","slug":"v_ldexp_f16","records":1,"summary":"Multiply the first input, a floating point value, by an integral power of 2 specified in the second input, a signed integer value, and store the…","page":"https://instructionsets.com/amdgpu/v_ldexp_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_ldexp_f16.json"},{"mnemonic":"v_ldexp_f16_fake16","slug":"v_ldexp_f16_fake16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_ldexp_f16_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_ldexp_f16_fake16.json"},{"mnemonic":"v_ldexp_f16_t16","slug":"v_ldexp_f16_t16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_ldexp_f16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_ldexp_f16_t16.json"},{"mnemonic":"v_ldexp_f32","slug":"v_ldexp_f32","records":1,"summary":"Multiply the first input, a floating point value, by an integral power of 2 specified in the second input, a signed integer value, and store the…","page":"https://instructionsets.com/amdgpu/v_ldexp_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_ldexp_f32.json"},{"mnemonic":"v_ldexp_f64","slug":"v_ldexp_f64","records":1,"summary":"Multiply the first input, a floating point value, by an integral power of 2 specified in the second input, a signed integer value, and store the…","page":"https://instructionsets.com/amdgpu/v_ldexp_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_ldexp_f64.json"},{"mnemonic":"v_lerp_u8","slug":"v_lerp_u8","records":1,"summary":"Average two 4-D vectors stored as packed bytes in the first two inputs with rounding control provided by the third input, then store the result into…","page":"https://instructionsets.com/amdgpu/v_lerp_u8/","api":"https://instructionsets.com/api/v1/amdgpu/v_lerp_u8.json"},{"mnemonic":"v_log_bf16","slug":"v_log_bf16","records":1,"summary":"AMDGPU VOP1 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_log_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_log_bf16.json"},{"mnemonic":"v_log_f16","slug":"v_log_f16","records":1,"summary":"Calculate the base 2 logarithm of the half-precision float input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_log_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_log_f16.json"},{"mnemonic":"v_log_f32","slug":"v_log_f32","records":1,"summary":"Calculate the base 2 logarithm of the single-precision float input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_log_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_log_f32.json"},{"mnemonic":"v_log_legacy_f32","slug":"v_log_legacy_f32","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_log_legacy_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_log_legacy_f32.json"},{"mnemonic":"v_lshl_add_u32","slug":"v_lshl_add_u32","records":1,"summary":"Given a shift count in the second input, calculate the logical shift left of the first input, then add the third input to the intermediate result…","page":"https://instructionsets.com/amdgpu/v_lshl_add_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshl_add_u32.json"},{"mnemonic":"v_lshl_add_u64","slug":"v_lshl_add_u64","records":1,"summary":"Given a shift count in the second input, calculate the logical shift left of the first input, then add the third input to the intermediate result…","page":"https://instructionsets.com/amdgpu/v_lshl_add_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshl_add_u64.json"},{"mnemonic":"v_lshl_b32","slug":"v_lshl_b32","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_lshl_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshl_b32.json"},{"mnemonic":"v_lshl_b64","slug":"v_lshl_b64","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_lshl_b64/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshl_b64.json"},{"mnemonic":"v_lshl_or_b32","slug":"v_lshl_or_b32","records":1,"summary":"Given a shift count in the second input, calculate the logical shift left of the first input, then calculate the bitwise OR of the intermediate…","page":"https://instructionsets.com/amdgpu/v_lshl_or_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshl_or_b32.json"},{"mnemonic":"v_lshlrev_b16","slug":"v_lshlrev_b16","records":1,"summary":"Given a shift count in the first vector input, calculate the logical shift left of the second vector input and store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_lshlrev_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshlrev_b16.json"},{"mnemonic":"v_lshlrev_b32","slug":"v_lshlrev_b32","records":1,"summary":"Given a shift count in the first vector input, calculate the logical shift left of the second vector input and store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_lshlrev_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshlrev_b32.json"},{"mnemonic":"v_lshlrev_b64","slug":"v_lshlrev_b64","records":1,"summary":"Given a shift count in the first vector input, calculate the logical shift left of the second vector input and store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_lshlrev_b64/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshlrev_b64.json"},{"mnemonic":"v_lshlrev_b64_pseudo","slug":"v_lshlrev_b64_pseudo","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_lshlrev_b64_pseudo/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshlrev_b64_pseudo.json"},{"mnemonic":"v_lshr_b32","slug":"v_lshr_b32","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_lshr_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshr_b32.json"},{"mnemonic":"v_lshr_b64","slug":"v_lshr_b64","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_lshr_b64/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshr_b64.json"},{"mnemonic":"v_lshrrev_b16","slug":"v_lshrrev_b16","records":1,"summary":"Given a shift count in the first vector input, calculate the logical shift right of the second vector input and store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_lshrrev_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshrrev_b16.json"},{"mnemonic":"v_lshrrev_b32","slug":"v_lshrrev_b32","records":1,"summary":"Given a shift count in the first vector input, calculate the logical shift right of the second vector input and store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_lshrrev_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshrrev_b32.json"},{"mnemonic":"v_lshrrev_b64","slug":"v_lshrrev_b64","records":1,"summary":"Given a shift count in the first vector input, calculate the logical shift right of the second vector input and store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_lshrrev_b64/","api":"https://instructionsets.com/api/v1/amdgpu/v_lshrrev_b64.json"},{"mnemonic":"v_mac_f16","slug":"v_mac_f16","records":1,"summary":"Multiply two floating point inputs and accumulate the result into the destination register. Implements IEEE rules and non-standard rule for OPSEL.","page":"https://instructionsets.com/amdgpu/v_mac_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mac_f16.json"},{"mnemonic":"v_mac_f32","slug":"v_mac_f32","records":1,"summary":"Multiply two floating point inputs and accumulate the result into the destination register.","page":"https://instructionsets.com/amdgpu/v_mac_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mac_f32.json"},{"mnemonic":"v_mac_legacy_f32","slug":"v_mac_legacy_f32","records":1,"summary":"Multiply and add single-precision values, accumulate with destination. Follows DX9 rules where 0.0 times anything produces 0.0.","page":"https://instructionsets.com/amdgpu/v_mac_legacy_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mac_legacy_f32.json"},{"mnemonic":"v_mad_co_i64_i32","slug":"v_mad_co_i64_i32","records":1,"summary":"Multiply two signed integer inputs, add a third signed integer input, store the result into a 64-bit vector register and store the overflow/carryout…","page":"https://instructionsets.com/amdgpu/v_mad_co_i64_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_co_i64_i32.json","aliases":["v_mad_i64_i32"]},{"mnemonic":"v_mad_co_u64_u32","slug":"v_mad_co_u64_u32","records":1,"summary":"Multiply two unsigned integer inputs, add a third unsigned integer input, store the result into a 64-bit vector register and store the…","page":"https://instructionsets.com/amdgpu/v_mad_co_u64_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_co_u64_u32.json","aliases":["v_mad_u64_u32"]},{"mnemonic":"v_mad_f16","slug":"v_mad_f16","records":1,"summary":"Multiply two half-precision float inputs and add a third input, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mad_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_f16.json"},{"mnemonic":"v_mad_f16_gfx9","slug":"v_mad_f16_gfx9","records":1,"summary":"AMDGPU VOP3 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mad_f16_gfx9/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_f16_gfx9.json"},{"mnemonic":"v_mad_f32","slug":"v_mad_f32","records":1,"summary":"Multiply two single-precision float inputs and add a third input, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mad_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_f32.json"},{"mnemonic":"v_mad_i16","slug":"v_mad_i16","records":1,"summary":"Multiply two signed 16-bit integer inputs, add a signed 16-bit integer value from a third input, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mad_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_i16.json"},{"mnemonic":"v_mad_i16_gfx9","slug":"v_mad_i16_gfx9","records":1,"summary":"AMDGPU VOP3 vector instruction operating on i16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mad_i16_gfx9/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_i16_gfx9.json"},{"mnemonic":"v_mad_i32_i16","slug":"v_mad_i32_i16","records":1,"summary":"Multiply two signed 16-bit integer inputs in the signed 32-bit integer domain, add a signed 32-bit integer value from a third input, and store the…","page":"https://instructionsets.com/amdgpu/v_mad_i32_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_i32_i16.json"},{"mnemonic":"v_mad_i32_i24","slug":"v_mad_i32_i24","records":1,"summary":"Multiply two signed 24-bit integer inputs in the signed 32-bit integer domain, add a signed 32-bit integer value from a third input, and store the…","page":"https://instructionsets.com/amdgpu/v_mad_i32_i24/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_i32_i24.json"},{"mnemonic":"v_mad_i64_i32","slug":"v_mad_i64_i32","records":1,"summary":"Multiply two signed integer inputs, add a third signed integer input, store the result into a 64-bit vector register and store the overflow/carryout…","page":"https://instructionsets.com/amdgpu/v_mad_i64_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_i64_i32.json","aliases":["v_mad_co_i64_i32"]},{"mnemonic":"v_mad_legacy_f16","slug":"v_mad_legacy_f16","records":1,"summary":"Multiply add of FP16 values. Implements IEEE rules and non-standard rule for OPSEL.","page":"https://instructionsets.com/amdgpu/v_mad_legacy_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_legacy_f16.json"},{"mnemonic":"v_mad_legacy_f32","slug":"v_mad_legacy_f32","records":1,"summary":"Multiply and add single-precision values. Follows DX9 rules where 0.0 times anything produces 0.0.","page":"https://instructionsets.com/amdgpu/v_mad_legacy_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_legacy_f32.json"},{"mnemonic":"v_mad_legacy_i16","slug":"v_mad_legacy_i16","records":1,"summary":"Multiply add of signed short values. Has non-standard rule for OPSEL.","page":"https://instructionsets.com/amdgpu/v_mad_legacy_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_legacy_i16.json"},{"mnemonic":"v_mad_legacy_u16","slug":"v_mad_legacy_u16","records":1,"summary":"Multiply add of unsigned short values. Has non-standard rule for OPSEL.","page":"https://instructionsets.com/amdgpu/v_mad_legacy_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_legacy_u16.json"},{"mnemonic":"v_mad_mix_f32","slug":"v_mad_mix_f32","records":1,"summary":"Multiply two inputs and add a third input where the inputs are a mix of half-precision float and single- precision float values.","page":"https://instructionsets.com/amdgpu/v_mad_mix_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_mix_f32.json"},{"mnemonic":"v_mad_mixhi_f16","slug":"v_mad_mixhi_f16","records":1,"summary":"Multiply two inputs and add a third input where the inputs are a mix of half-precision float and single- precision float values.","page":"https://instructionsets.com/amdgpu/v_mad_mixhi_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_mixhi_f16.json"},{"mnemonic":"v_mad_mixlo_f16","slug":"v_mad_mixlo_f16","records":1,"summary":"Multiply two inputs and add a third input where the inputs are a mix of half-precision float and single- precision float values.","page":"https://instructionsets.com/amdgpu/v_mad_mixlo_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_mixlo_f16.json"},{"mnemonic":"v_mad_nc_i64_i32","slug":"v_mad_nc_i64_i32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on i32/i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mad_nc_i64_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_nc_i64_i32.json"},{"mnemonic":"v_mad_nc_u64_u32","slug":"v_mad_nc_u64_u32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on u32/u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mad_nc_u64_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_nc_u64_u32.json"},{"mnemonic":"v_mad_u16","slug":"v_mad_u16","records":1,"summary":"Multiply two unsigned 16-bit integer inputs, add an unsigned 16-bit integer value from a third input, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mad_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_u16.json"},{"mnemonic":"v_mad_u16_gfx9","slug":"v_mad_u16_gfx9","records":1,"summary":"AMDGPU VOP3 vector instruction operating on u16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mad_u16_gfx9/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_u16_gfx9.json"},{"mnemonic":"v_mad_u32","slug":"v_mad_u32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mad_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_u32.json"},{"mnemonic":"v_mad_u32_u16","slug":"v_mad_u32_u16","records":1,"summary":"Multiply two unsigned 16-bit integer inputs in the unsigned 32-bit integer domain, add an unsigned 32-bit integer value from a third input, and store…","page":"https://instructionsets.com/amdgpu/v_mad_u32_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_u32_u16.json"},{"mnemonic":"v_mad_u32_u24","slug":"v_mad_u32_u24","records":1,"summary":"Multiply two unsigned 24-bit integer inputs in the unsigned 32-bit integer domain, add a unsigned 32-bit integer value from a third input, and store…","page":"https://instructionsets.com/amdgpu/v_mad_u32_u24/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_u32_u24.json"},{"mnemonic":"v_mad_u64_u32","slug":"v_mad_u64_u32","records":1,"summary":"Multiply two unsigned integer inputs, add a third unsigned integer input, store the result into a 64-bit vector register and store the…","page":"https://instructionsets.com/amdgpu/v_mad_u64_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mad_u64_u32.json","aliases":["v_mad_co_u64_u32"]},{"mnemonic":"v_madak_f16","slug":"v_madak_f16","records":1,"summary":"Multiply two floating point inputs and add a literal constant, and store the result into a vector register. Implements IEEE rules.","page":"https://instructionsets.com/amdgpu/v_madak_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_madak_f16.json"},{"mnemonic":"v_madak_f32","slug":"v_madak_f32","records":1,"summary":"Multiply two floating point inputs and add a literal constant, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_madak_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_madak_f32.json"},{"mnemonic":"v_madmk_f16","slug":"v_madmk_f16","records":1,"summary":"Multiply a floating point input with a literal constant and add a second floating point input, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_madmk_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_madmk_f16.json"},{"mnemonic":"v_madmk_f32","slug":"v_madmk_f32","records":1,"summary":"Multiply a floating point input with a literal constant and add a second floating point input, and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_madmk_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_madmk_f32.json"},{"mnemonic":"v_max3_f16","slug":"v_max3_f16","records":1,"summary":"Select the maximum of three half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max3_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_max3_f16.json","aliases":["v_max3_num_f16"]},{"mnemonic":"v_max3_f32","slug":"v_max3_f32","records":1,"summary":"Select the maximum of three single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max3_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_max3_f32.json","aliases":["v_max3_num_f32"]},{"mnemonic":"v_max3_i16","slug":"v_max3_i16","records":1,"summary":"Select the maximum of three signed 16-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max3_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_max3_i16.json"},{"mnemonic":"v_max3_i32","slug":"v_max3_i32","records":1,"summary":"Select the maximum of three signed 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max3_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_max3_i32.json"},{"mnemonic":"v_max3_num_f16","slug":"v_max3_num_f16","records":1,"summary":"Select the IEEE maximumNumber() of three half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max3_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_max3_num_f16.json","aliases":["v_max3_f16"]},{"mnemonic":"v_max3_num_f32","slug":"v_max3_num_f32","records":1,"summary":"Select the IEEE maximumNumber() of three single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max3_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_max3_num_f32.json","aliases":["v_max3_f32"]},{"mnemonic":"v_max3_u16","slug":"v_max3_u16","records":1,"summary":"Select the maximum of three unsigned 16-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max3_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_max3_u16.json"},{"mnemonic":"v_max3_u32","slug":"v_max3_u32","records":1,"summary":"Select the maximum of three unsigned 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max3_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_max3_u32.json"},{"mnemonic":"v_max_bf16","slug":"v_max_bf16","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_max_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_bf16.json"},{"mnemonic":"v_max_f16","slug":"v_max_f16","records":1,"summary":"Select the maximum of two half-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_f16.json","aliases":["v_max_num_f16"]},{"mnemonic":"v_max_f32","slug":"v_max_f32","records":1,"summary":"Select the maximum of two single-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_f32.json","aliases":["v_max_num_f32"]},{"mnemonic":"v_max_f64","slug":"v_max_f64","records":1,"summary":"Select the maximum of two double-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_f64.json","aliases":["v_max_num_f64"]},{"mnemonic":"v_max_i16","slug":"v_max_i16","records":1,"summary":"Select the maximum of two signed 16-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_i16.json"},{"mnemonic":"v_max_i32","slug":"v_max_i32","records":1,"summary":"Select the maximum of two signed 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_i32.json"},{"mnemonic":"v_max_i64","slug":"v_max_i64","records":1,"summary":"AMDGPU VOP3 vector instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_max_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_i64.json"},{"mnemonic":"v_max_legacy_f32","slug":"v_max_legacy_f32","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_max_legacy_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_legacy_f32.json"},{"mnemonic":"v_max_num_f16","slug":"v_max_num_f16","records":1,"summary":"Select the IEEE maximumNumber() of two half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_num_f16.json","aliases":["v_max_f16"]},{"mnemonic":"v_max_num_f32","slug":"v_max_num_f32","records":1,"summary":"Select the IEEE maximumNumber() of two single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_num_f32.json","aliases":["v_max_f32"]},{"mnemonic":"v_max_num_f64","slug":"v_max_num_f64","records":1,"summary":"Select the IEEE maximumNumber() of two double-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_num_f64.json","aliases":["v_max_f64"]},{"mnemonic":"v_max_u16","slug":"v_max_u16","records":1,"summary":"Select the maximum of two unsigned 16-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_u16.json"},{"mnemonic":"v_max_u32","slug":"v_max_u32","records":1,"summary":"Select the maximum of two unsigned 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_max_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_u32.json"},{"mnemonic":"v_max_u64","slug":"v_max_u64","records":1,"summary":"AMDGPU VOP3 vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_max_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_max_u64.json"},{"mnemonic":"v_maximum3_f16","slug":"v_maximum3_f16","records":1,"summary":"Select the IEEE maximum() of three half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_maximum3_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_maximum3_f16.json"},{"mnemonic":"v_maximum3_f32","slug":"v_maximum3_f32","records":1,"summary":"Select the IEEE maximum() of three single-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_maximum3_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_maximum3_f32.json"},{"mnemonic":"v_maximum_f16","slug":"v_maximum_f16","records":1,"summary":"Select the IEEE maximum() of two half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_maximum_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_maximum_f16.json"},{"mnemonic":"v_maximum_f32","slug":"v_maximum_f32","records":1,"summary":"Select the IEEE maximum() of two single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_maximum_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_maximum_f32.json"},{"mnemonic":"v_maximum_f64","slug":"v_maximum_f64","records":1,"summary":"Select the IEEE maximum() of two double-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_maximum_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_maximum_f64.json"},{"mnemonic":"v_maximumminimum_f16","slug":"v_maximumminimum_f16","records":1,"summary":"Select the IEEE maximum() of the first two half-precision float inputs and then select the IEEE minimum() of that result and third half-precision…","page":"https://instructionsets.com/amdgpu/v_maximumminimum_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_maximumminimum_f16.json"},{"mnemonic":"v_maximumminimum_f32","slug":"v_maximumminimum_f32","records":1,"summary":"Select the IEEE maximum() of the first two single-precision float inputs and then select the IEEE minimum() of that result and third single-precision…","page":"https://instructionsets.com/amdgpu/v_maximumminimum_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_maximumminimum_f32.json"},{"mnemonic":"v_maxmin_f16","slug":"v_maxmin_f16","records":1,"summary":"Select the maximum of the first two half-precision float inputs and then select the minimum of that result and third half-precision float input.","page":"https://instructionsets.com/amdgpu/v_maxmin_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_maxmin_f16.json","aliases":["v_maxmin_num_f16"]},{"mnemonic":"v_maxmin_f32","slug":"v_maxmin_f32","records":1,"summary":"Select the maximum of the first two single-precision float inputs and then select the minimum of that result and third single-precision float input.","page":"https://instructionsets.com/amdgpu/v_maxmin_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_maxmin_f32.json","aliases":["v_maxmin_num_f32"]},{"mnemonic":"v_maxmin_i32","slug":"v_maxmin_i32","records":1,"summary":"Select the maximum of the first two signed 32-bit integer inputs and then select the minimum of that result and third signed 32-bit integer input.","page":"https://instructionsets.com/amdgpu/v_maxmin_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_maxmin_i32.json"},{"mnemonic":"v_maxmin_num_f16","slug":"v_maxmin_num_f16","records":1,"summary":"Select the IEEE maximumNumber() of the first two half-precision float inputs and then select the IEEE minimumNumber() of that result and third…","page":"https://instructionsets.com/amdgpu/v_maxmin_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_maxmin_num_f16.json","aliases":["v_maxmin_f16"]},{"mnemonic":"v_maxmin_num_f32","slug":"v_maxmin_num_f32","records":1,"summary":"Select the IEEE maximumNumber() of the first two single-precision float inputs and then select the IEEE minimumNumber() of that result and third…","page":"https://instructionsets.com/amdgpu/v_maxmin_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_maxmin_num_f32.json","aliases":["v_maxmin_f32"]},{"mnemonic":"v_maxmin_u32","slug":"v_maxmin_u32","records":1,"summary":"Select the maximum of the first two unsigned 32-bit integer inputs and then select the minimum of that result and third unsigned 32-bit integer input.","page":"https://instructionsets.com/amdgpu/v_maxmin_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_maxmin_u32.json"},{"mnemonic":"v_mbcnt_hi_u32_b32","slug":"v_mbcnt_hi_u32_b32","records":1,"summary":"For each lane 32 <= N < 64, examine the N least significant bits of the first input and count how many of those bits are \"1\".","page":"https://instructionsets.com/amdgpu/v_mbcnt_hi_u32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mbcnt_hi_u32_b32.json"},{"mnemonic":"v_mbcnt_lo_u32_b32","slug":"v_mbcnt_lo_u32_b32","records":1,"summary":"For each lane 0 <= N < 32, examine the N least significant bits of the first input and count how many of those bits are \"1\".","page":"https://instructionsets.com/amdgpu/v_mbcnt_lo_u32_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mbcnt_lo_u32_b32.json"},{"mnemonic":"v_med3_f16","slug":"v_med3_f16","records":1,"summary":"Select the median of three half-precision float values and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_med3_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_med3_f16.json","aliases":["v_med3_num_f16"]},{"mnemonic":"v_med3_f32","slug":"v_med3_f32","records":1,"summary":"Select the median of three single-precision float values and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_med3_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_med3_f32.json","aliases":["v_med3_num_f32"]},{"mnemonic":"v_med3_i16","slug":"v_med3_i16","records":1,"summary":"Select the median of three signed 16-bit integer values and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_med3_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_med3_i16.json"},{"mnemonic":"v_med3_i32","slug":"v_med3_i32","records":1,"summary":"Select the median of three signed 32-bit integer values and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_med3_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_med3_i32.json"},{"mnemonic":"v_med3_num_f16","slug":"v_med3_num_f16","records":1,"summary":"Select the median of three half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_med3_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_med3_num_f16.json","aliases":["v_med3_f16"]},{"mnemonic":"v_med3_num_f32","slug":"v_med3_num_f32","records":1,"summary":"Select the median of three single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_med3_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_med3_num_f32.json","aliases":["v_med3_f32"]},{"mnemonic":"v_med3_u16","slug":"v_med3_u16","records":1,"summary":"Select the median of three unsigned 16-bit integer values and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_med3_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_med3_u16.json"},{"mnemonic":"v_med3_u32","slug":"v_med3_u32","records":1,"summary":"Select the median of three unsigned 32-bit integer values and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_med3_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_med3_u32.json"},{"mnemonic":"v_mfma_f32_16x16x128_f8f6f4","slug":"v_mfma_f32_16x16x128_f8f6f4","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x128_f8f6f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x128_f8f6f4.json"},{"mnemonic":"v_mfma_f32_16x16x16_bf16","slug":"v_mfma_f32_16x16x16_bf16","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x16_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x16_bf16.json","aliases":["v_mfma_f32_16x16x16bf16","v_mfma_f32_16x16x16bf16_1k"]},{"mnemonic":"v_mfma_f32_16x16x16_f16","slug":"v_mfma_f32_16x16x16_f16","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x16_f16.json","aliases":["v_mfma_f32_16x16x16f16"]},{"mnemonic":"v_mfma_f32_16x16x16bf16_1k","slug":"v_mfma_f32_16x16x16bf16_1k","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x16bf16_1k/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x16bf16_1k.json","aliases":["v_mfma_f32_16x16x16_bf16","v_mfma_f32_16x16x16bf16"]},{"mnemonic":"v_mfma_f32_16x16x16f16","slug":"v_mfma_f32_16x16x16f16","records":1,"summary":"Matrix-fused-multiply-add: cooperative 16x16x16 matrix-multiply-accumulate on matrix-core hardware, fp16 inputs, fp32 accumulate.","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x16f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x16f16.json","aliases":["v_mfma_f32_16x16x16_f16"]},{"mnemonic":"v_mfma_f32_16x16x1_4b_f32","slug":"v_mfma_f32_16x16x1_4b_f32","records":1,"summary":"Multiply the 16x1 matrix in the first input by the 1x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x1_4b_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x1_4b_f32.json","aliases":["v_mfma_f32_16x16x1f32"]},{"mnemonic":"v_mfma_f32_16x16x1f32","slug":"v_mfma_f32_16x16x1f32","records":1,"summary":"Multiply the 16x1 matrix in the first input by the 1x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x1f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x1f32.json","aliases":["v_mfma_f32_16x16x1_4b_f32"]},{"mnemonic":"v_mfma_f32_16x16x2bf16","slug":"v_mfma_f32_16x16x2bf16","records":1,"summary":"Multiply the 16x2 matrix in the first input by the 2x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x2bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x2bf16.json"},{"mnemonic":"v_mfma_f32_16x16x32_bf16","slug":"v_mfma_f32_16x16x32_bf16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x32_bf16.json"},{"mnemonic":"v_mfma_f32_16x16x32_bf8_bf8","slug":"v_mfma_f32_16x16x32_bf8_bf8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x32_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x32_bf8_bf8.json","aliases":["v_mfma_f32_16x16x32bf8bf8"]},{"mnemonic":"v_mfma_f32_16x16x32_bf8_fp8","slug":"v_mfma_f32_16x16x32_bf8_fp8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x32_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x32_bf8_fp8.json","aliases":["v_mfma_f32_16x16x32bf8fp8"]},{"mnemonic":"v_mfma_f32_16x16x32_f16","slug":"v_mfma_f32_16x16x32_f16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x32_f16.json"},{"mnemonic":"v_mfma_f32_16x16x32_fp8_bf8","slug":"v_mfma_f32_16x16x32_fp8_bf8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x32_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x32_fp8_bf8.json","aliases":["v_mfma_f32_16x16x32fp8bf8"]},{"mnemonic":"v_mfma_f32_16x16x32_fp8_fp8","slug":"v_mfma_f32_16x16x32_fp8_fp8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x32_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x32_fp8_fp8.json","aliases":["v_mfma_f32_16x16x32fp8fp8"]},{"mnemonic":"v_mfma_f32_16x16x4_4b_bf16","slug":"v_mfma_f32_16x16x4_4b_bf16","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x4_4b_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x4_4b_bf16.json","aliases":["v_mfma_f32_16x16x4bf16","v_mfma_f32_16x16x4bf16_1k"]},{"mnemonic":"v_mfma_f32_16x16x4_4b_f16","slug":"v_mfma_f32_16x16x4_4b_f16","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x4_4b_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x4_4b_f16.json","aliases":["v_mfma_f32_16x16x4f16"]},{"mnemonic":"v_mfma_f32_16x16x4_f32","slug":"v_mfma_f32_16x16x4_f32","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x4_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x4_f32.json","aliases":["v_mfma_f32_16x16x4f32"]},{"mnemonic":"v_mfma_f32_16x16x4bf16_1k","slug":"v_mfma_f32_16x16x4bf16_1k","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x4bf16_1k/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x4bf16_1k.json","aliases":["v_mfma_f32_16x16x4_4b_bf16","v_mfma_f32_16x16x4bf16"]},{"mnemonic":"v_mfma_f32_16x16x4f16","slug":"v_mfma_f32_16x16x4f16","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x4f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x4f16.json","aliases":["v_mfma_f32_16x16x4_4b_f16"]},{"mnemonic":"v_mfma_f32_16x16x4f32","slug":"v_mfma_f32_16x16x4f32","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x4f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x4f32.json","aliases":["v_mfma_f32_16x16x4_f32"]},{"mnemonic":"v_mfma_f32_16x16x8_xf32","slug":"v_mfma_f32_16x16x8_xf32","records":1,"summary":"Multiply the 16x8 matrix in the first input by the 8x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x8_xf32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x8_xf32.json","aliases":["v_mfma_f32_16x16x8xf32"]},{"mnemonic":"v_mfma_f32_16x16x8bf16","slug":"v_mfma_f32_16x16x8bf16","records":1,"summary":"Multiply the 16x8 matrix in the first input by the 8x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_16x16x8bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_16x16x8bf16.json"},{"mnemonic":"v_mfma_f32_32x32x16_bf16","slug":"v_mfma_f32_32x32x16_bf16","records":1,"summary":"Multiply the 32x16 matrix in the first input by the 16x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x16_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x16_bf16.json"},{"mnemonic":"v_mfma_f32_32x32x16_bf8_bf8","slug":"v_mfma_f32_32x32x16_bf8_bf8","records":1,"summary":"Multiply the 32x16 matrix in the first input by the 16x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x16_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x16_bf8_bf8.json","aliases":["v_mfma_f32_32x32x16bf8bf8"]},{"mnemonic":"v_mfma_f32_32x32x16_bf8_fp8","slug":"v_mfma_f32_32x32x16_bf8_fp8","records":1,"summary":"Multiply the 32x16 matrix in the first input by the 16x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x16_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x16_bf8_fp8.json","aliases":["v_mfma_f32_32x32x16bf8fp8"]},{"mnemonic":"v_mfma_f32_32x32x16_f16","slug":"v_mfma_f32_32x32x16_f16","records":1,"summary":"Multiply the 32x16 matrix in the first input by the 16x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x16_f16.json"},{"mnemonic":"v_mfma_f32_32x32x16_fp8_bf8","slug":"v_mfma_f32_32x32x16_fp8_bf8","records":1,"summary":"Multiply the 32x16 matrix in the first input by the 16x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x16_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x16_fp8_bf8.json","aliases":["v_mfma_f32_32x32x16fp8bf8"]},{"mnemonic":"v_mfma_f32_32x32x16_fp8_fp8","slug":"v_mfma_f32_32x32x16_fp8_fp8","records":1,"summary":"Multiply the 32x16 matrix in the first input by the 16x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x16_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x16_fp8_fp8.json","aliases":["v_mfma_f32_32x32x16fp8fp8"]},{"mnemonic":"v_mfma_f32_32x32x1_2b_f32","slug":"v_mfma_f32_32x32x1_2b_f32","records":1,"summary":"Multiply the 32x1 matrix in the first input by the 1x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x1_2b_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x1_2b_f32.json","aliases":["v_mfma_f32_32x32x1f32"]},{"mnemonic":"v_mfma_f32_32x32x1f32","slug":"v_mfma_f32_32x32x1f32","records":1,"summary":"Multiply the 32x1 matrix in the first input by the 1x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x1f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x1f32.json","aliases":["v_mfma_f32_32x32x1_2b_f32"]},{"mnemonic":"v_mfma_f32_32x32x2_f32","slug":"v_mfma_f32_32x32x2_f32","records":1,"summary":"Multiply the 32x2 matrix in the first input by the 2x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x2_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x2_f32.json","aliases":["v_mfma_f32_32x32x2f32"]},{"mnemonic":"v_mfma_f32_32x32x2bf16","slug":"v_mfma_f32_32x32x2bf16","records":1,"summary":"Multiply the 32x2 matrix in the first input by the 2x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x2bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x2bf16.json"},{"mnemonic":"v_mfma_f32_32x32x2f32","slug":"v_mfma_f32_32x32x2f32","records":1,"summary":"Multiply the 32x2 matrix in the first input by the 2x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x2f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x2f32.json","aliases":["v_mfma_f32_32x32x2_f32"]},{"mnemonic":"v_mfma_f32_32x32x4_2b_bf16","slug":"v_mfma_f32_32x32x4_2b_bf16","records":1,"summary":"Multiply the 32x4 matrix in the first input by the 4x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x4_2b_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x4_2b_bf16.json","aliases":["v_mfma_f32_32x32x4bf16_1k"]},{"mnemonic":"v_mfma_f32_32x32x4_2b_f16","slug":"v_mfma_f32_32x32x4_2b_f16","records":1,"summary":"Multiply the 32x4 matrix in the first input by the 4x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x4_2b_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x4_2b_f16.json","aliases":["v_mfma_f32_32x32x4f16"]},{"mnemonic":"v_mfma_f32_32x32x4_xf32","slug":"v_mfma_f32_32x32x4_xf32","records":1,"summary":"Multiply the 32x4 matrix in the first input by the 4x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x4_xf32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x4_xf32.json","aliases":["v_mfma_f32_32x32x4xf32"]},{"mnemonic":"v_mfma_f32_32x32x4bf16","slug":"v_mfma_f32_32x32x4bf16","records":1,"summary":"Multiply the 32x4 matrix in the first input by the 4x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x4bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x4bf16.json"},{"mnemonic":"v_mfma_f32_32x32x4bf16_1k","slug":"v_mfma_f32_32x32x4bf16_1k","records":1,"summary":"Multiply the 32x4 matrix in the first input by the 4x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x4bf16_1k/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x4bf16_1k.json","aliases":["v_mfma_f32_32x32x4_2b_bf16"]},{"mnemonic":"v_mfma_f32_32x32x4f16","slug":"v_mfma_f32_32x32x4f16","records":1,"summary":"Multiply the 32x4 matrix in the first input by the 4x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x4f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x4f16.json","aliases":["v_mfma_f32_32x32x4_2b_f16"]},{"mnemonic":"v_mfma_f32_32x32x64_f8f6f4","slug":"v_mfma_f32_32x32x64_f8f6f4","records":1,"summary":"Multiply the 32x64 matrix in the first input by the 64x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x64_f8f6f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x64_f8f6f4.json"},{"mnemonic":"v_mfma_f32_32x32x8_bf16","slug":"v_mfma_f32_32x32x8_bf16","records":1,"summary":"Multiply the 32x8 matrix in the first input by the 8x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x8_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x8_bf16.json","aliases":["v_mfma_f32_32x32x8bf16","v_mfma_f32_32x32x8bf16_1k"]},{"mnemonic":"v_mfma_f32_32x32x8_f16","slug":"v_mfma_f32_32x32x8_f16","records":1,"summary":"Multiply the 32x8 matrix in the first input by the 8x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x8_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x8_f16.json","aliases":["v_mfma_f32_32x32x8f16"]},{"mnemonic":"v_mfma_f32_32x32x8bf16_1k","slug":"v_mfma_f32_32x32x8bf16_1k","records":1,"summary":"Multiply the 32x8 matrix in the first input by the 8x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x8bf16_1k/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x8bf16_1k.json","aliases":["v_mfma_f32_32x32x8_bf16","v_mfma_f32_32x32x8bf16"]},{"mnemonic":"v_mfma_f32_32x32x8f16","slug":"v_mfma_f32_32x32x8f16","records":1,"summary":"Multiply the 32x8 matrix in the first input by the 8x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f32_32x32x8f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_32x32x8f16.json","aliases":["v_mfma_f32_32x32x8_f16"]},{"mnemonic":"v_mfma_f32_4x4x1_16b_f32","slug":"v_mfma_f32_4x4x1_16b_f32","records":1,"summary":"Multiply the 4x1 matrix in the first input by the 1x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_f32_4x4x1_16b_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_4x4x1_16b_f32.json","aliases":["v_mfma_f32_4x4x1f32"]},{"mnemonic":"v_mfma_f32_4x4x1f32","slug":"v_mfma_f32_4x4x1f32","records":1,"summary":"Multiply the 4x1 matrix in the first input by the 1x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_f32_4x4x1f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_4x4x1f32.json","aliases":["v_mfma_f32_4x4x1_16b_f32"]},{"mnemonic":"v_mfma_f32_4x4x2bf16","slug":"v_mfma_f32_4x4x2bf16","records":1,"summary":"Multiply the 4x2 matrix in the first input by the 2x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_f32_4x4x2bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_4x4x2bf16.json"},{"mnemonic":"v_mfma_f32_4x4x4_16b_bf16","slug":"v_mfma_f32_4x4x4_16b_bf16","records":1,"summary":"Multiply the 4x4 matrix in the first input by the 4x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_f32_4x4x4_16b_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_4x4x4_16b_bf16.json","aliases":["v_mfma_f32_4x4x4bf16","v_mfma_f32_4x4x4bf16_1k"]},{"mnemonic":"v_mfma_f32_4x4x4_16b_f16","slug":"v_mfma_f32_4x4x4_16b_f16","records":1,"summary":"Multiply the 4x4 matrix in the first input by the 4x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_f32_4x4x4_16b_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_4x4x4_16b_f16.json","aliases":["v_mfma_f32_4x4x4f16"]},{"mnemonic":"v_mfma_f32_4x4x4bf16_1k","slug":"v_mfma_f32_4x4x4bf16_1k","records":1,"summary":"Multiply the 4x4 matrix in the first input by the 4x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_f32_4x4x4bf16_1k/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_4x4x4bf16_1k.json","aliases":["v_mfma_f32_4x4x4_16b_bf16","v_mfma_f32_4x4x4bf16"]},{"mnemonic":"v_mfma_f32_4x4x4f16","slug":"v_mfma_f32_4x4x4f16","records":1,"summary":"Multiply the 4x4 matrix in the first input by the 4x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_f32_4x4x4f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f32_4x4x4f16.json","aliases":["v_mfma_f32_4x4x4_16b_f16"]},{"mnemonic":"v_mfma_f64_16x16x4_f64","slug":"v_mfma_f64_16x16x4_f64","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f64_16x16x4_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f64_16x16x4_f64.json","aliases":["v_mfma_f64_16x16x4f64"]},{"mnemonic":"v_mfma_f64_16x16x4f64","slug":"v_mfma_f64_16x16x4f64","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_f64_16x16x4f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f64_16x16x4f64.json","aliases":["v_mfma_f64_16x16x4_f64"]},{"mnemonic":"v_mfma_f64_4x4x4_4b_f64","slug":"v_mfma_f64_4x4x4_4b_f64","records":1,"summary":"Multiply the 4x4 matrix in the first input by the 4x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_f64_4x4x4_4b_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f64_4x4x4_4b_f64.json","aliases":["v_mfma_f64_4x4x4f64"]},{"mnemonic":"v_mfma_f64_4x4x4f64","slug":"v_mfma_f64_4x4x4f64","records":1,"summary":"Multiply the 4x4 matrix in the first input by the 4x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_f64_4x4x4f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_f64_4x4x4f64.json","aliases":["v_mfma_f64_4x4x4_4b_f64"]},{"mnemonic":"v_mfma_i32_16x16x16i8","slug":"v_mfma_i32_16x16x16i8","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_16x16x16i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_16x16x16i8.json"},{"mnemonic":"v_mfma_i32_16x16x32_i8","slug":"v_mfma_i32_16x16x32_i8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_16x16x32_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_16x16x32_i8.json","aliases":["v_mfma_i32_16x16x32i8"]},{"mnemonic":"v_mfma_i32_16x16x4_4b_i8","slug":"v_mfma_i32_16x16x4_4b_i8","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_16x16x4_4b_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_16x16x4_4b_i8.json","aliases":["v_mfma_i32_16x16x4i8"]},{"mnemonic":"v_mfma_i32_16x16x4i8","slug":"v_mfma_i32_16x16x4i8","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_16x16x4i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_16x16x4i8.json","aliases":["v_mfma_i32_16x16x4_4b_i8"]},{"mnemonic":"v_mfma_i32_16x16x64_i8","slug":"v_mfma_i32_16x16x64_i8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_16x16x64_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_16x16x64_i8.json"},{"mnemonic":"v_mfma_i32_32x32x16_i8","slug":"v_mfma_i32_32x32x16_i8","records":1,"summary":"Multiply the 32x16 matrix in the first input by the 16x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_32x32x16_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_32x32x16_i8.json","aliases":["v_mfma_i32_32x32x16i8"]},{"mnemonic":"v_mfma_i32_32x32x32_i8","slug":"v_mfma_i32_32x32x32_i8","records":1,"summary":"Multiply the 32x32 matrix in the first input by the 32x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_32x32x32_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_32x32x32_i8.json"},{"mnemonic":"v_mfma_i32_32x32x4_2b_i8","slug":"v_mfma_i32_32x32x4_2b_i8","records":1,"summary":"Multiply the 32x4 matrix in the first input by the 4x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_32x32x4_2b_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_32x32x4_2b_i8.json","aliases":["v_mfma_i32_32x32x4i8"]},{"mnemonic":"v_mfma_i32_32x32x4i8","slug":"v_mfma_i32_32x32x4i8","records":1,"summary":"Multiply the 32x4 matrix in the first input by the 4x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_32x32x4i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_32x32x4i8.json","aliases":["v_mfma_i32_32x32x4_2b_i8"]},{"mnemonic":"v_mfma_i32_32x32x8i8","slug":"v_mfma_i32_32x32x8i8","records":1,"summary":"Multiply the 32x8 matrix in the first input by the 8x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_i32_32x32x8i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_32x32x8i8.json"},{"mnemonic":"v_mfma_i32_4x4x4_16b_i8","slug":"v_mfma_i32_4x4x4_16b_i8","records":1,"summary":"Multiply the 4x4 matrix in the first input by the 4x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_i32_4x4x4_16b_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_4x4x4_16b_i8.json","aliases":["v_mfma_i32_4x4x4i8"]},{"mnemonic":"v_mfma_i32_4x4x4i8","slug":"v_mfma_i32_4x4x4i8","records":1,"summary":"Multiply the 4x4 matrix in the first input by the 4x4 matrix in the second input and add the 4x4 matrix in the third input using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_mfma_i32_4x4x4i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_i32_4x4x4i8.json","aliases":["v_mfma_i32_4x4x4_16b_i8"]},{"mnemonic":"v_mfma_ld_scale_b32","slug":"v_mfma_ld_scale_b32","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mfma_ld_scale_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_ld_scale_b32.json"},{"mnemonic":"v_mfma_scale_f32_16x16x128_f8f6f4","slug":"v_mfma_scale_f32_16x16x128_f8f6f4","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_mfma_scale_f32_16x16x128_f8f6f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_scale_f32_16x16x128_f8f6f4.json"},{"mnemonic":"v_mfma_scale_f32_32x32x64_f8f6f4","slug":"v_mfma_scale_f32_32x32x64_f8f6f4","records":1,"summary":"Multiply the 32x64 matrix in the first input by the 64x32 matrix in the second input and add the 32x32 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_mfma_scale_f32_32x32x64_f8f6f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_mfma_scale_f32_32x32x64_f8f6f4.json"},{"mnemonic":"v_min3_f16","slug":"v_min3_f16","records":1,"summary":"Select the minimum of three half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min3_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_min3_f16.json","aliases":["v_min3_num_f16"]},{"mnemonic":"v_min3_f32","slug":"v_min3_f32","records":1,"summary":"Select the minimum of three single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min3_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_min3_f32.json","aliases":["v_min3_num_f32"]},{"mnemonic":"v_min3_i16","slug":"v_min3_i16","records":1,"summary":"Select the minimum of three signed 16-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min3_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_min3_i16.json"},{"mnemonic":"v_min3_i32","slug":"v_min3_i32","records":1,"summary":"Select the minimum of three signed 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min3_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_min3_i32.json"},{"mnemonic":"v_min3_num_f16","slug":"v_min3_num_f16","records":1,"summary":"Select the IEEE minimumNumber() of three half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min3_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_min3_num_f16.json","aliases":["v_min3_f16"]},{"mnemonic":"v_min3_num_f32","slug":"v_min3_num_f32","records":1,"summary":"Select the IEEE minimumNumber() of three single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min3_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_min3_num_f32.json","aliases":["v_min3_f32"]},{"mnemonic":"v_min3_u16","slug":"v_min3_u16","records":1,"summary":"Select the minimum of three unsigned 16-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min3_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_min3_u16.json"},{"mnemonic":"v_min3_u32","slug":"v_min3_u32","records":1,"summary":"Select the minimum of three unsigned 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min3_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_min3_u32.json"},{"mnemonic":"v_min_f16","slug":"v_min_f16","records":1,"summary":"Select the minimum of two half-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_f16.json","aliases":["v_min_num_f16"]},{"mnemonic":"v_min_f32","slug":"v_min_f32","records":1,"summary":"Select the minimum of two single-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_f32.json","aliases":["v_min_num_f32"]},{"mnemonic":"v_min_f64","slug":"v_min_f64","records":1,"summary":"Select the minimum of two double-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_f64.json","aliases":["v_min_num_f64"]},{"mnemonic":"v_min_i16","slug":"v_min_i16","records":1,"summary":"Select the minimum of two signed 16-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_i16.json"},{"mnemonic":"v_min_i32","slug":"v_min_i32","records":1,"summary":"Select the minimum of two signed 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_i32.json"},{"mnemonic":"v_min_i64","slug":"v_min_i64","records":1,"summary":"AMDGPU VOP3 vector instruction operating on i64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_min_i64/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_i64.json"},{"mnemonic":"v_min_legacy_f32","slug":"v_min_legacy_f32","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_min_legacy_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_legacy_f32.json"},{"mnemonic":"v_min_num_f16","slug":"v_min_num_f16","records":1,"summary":"Select the IEEE minimumNumber() of two half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_num_f16.json","aliases":["v_min_f16"]},{"mnemonic":"v_min_num_f32","slug":"v_min_num_f32","records":1,"summary":"Select the IEEE minimumNumber() of two single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_num_f32.json","aliases":["v_min_f32"]},{"mnemonic":"v_min_num_f64","slug":"v_min_num_f64","records":1,"summary":"Select the IEEE minimumNumber() of two double-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_num_f64.json","aliases":["v_min_f64"]},{"mnemonic":"v_min_u16","slug":"v_min_u16","records":1,"summary":"Select the minimum of two unsigned 16-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_u16.json"},{"mnemonic":"v_min_u32","slug":"v_min_u32","records":1,"summary":"Select the minimum of two unsigned 32-bit integer inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_min_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_u32.json"},{"mnemonic":"v_min_u64","slug":"v_min_u64","records":1,"summary":"AMDGPU VOP3 vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_min_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_min_u64.json"},{"mnemonic":"v_minimum3_f16","slug":"v_minimum3_f16","records":1,"summary":"Select the IEEE minimum() of three half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_minimum3_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_minimum3_f16.json"},{"mnemonic":"v_minimum3_f32","slug":"v_minimum3_f32","records":1,"summary":"Select the IEEE minimum() of three single-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_minimum3_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_minimum3_f32.json"},{"mnemonic":"v_minimum_f16","slug":"v_minimum_f16","records":1,"summary":"Select the IEEE minimum() of two half-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_minimum_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_minimum_f16.json"},{"mnemonic":"v_minimum_f32","slug":"v_minimum_f32","records":1,"summary":"Select the IEEE minimum() of two single-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_minimum_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_minimum_f32.json"},{"mnemonic":"v_minimum_f64","slug":"v_minimum_f64","records":1,"summary":"Select the IEEE minimum() of two double-precision float inputs and store the selected value into a vector register.","page":"https://instructionsets.com/amdgpu/v_minimum_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_minimum_f64.json"},{"mnemonic":"v_minimummaximum_f16","slug":"v_minimummaximum_f16","records":1,"summary":"Select the IEEE minimum() of the first two half-precision float inputs and then select the IEEE maximum() of that result and third half-precision…","page":"https://instructionsets.com/amdgpu/v_minimummaximum_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_minimummaximum_f16.json"},{"mnemonic":"v_minimummaximum_f32","slug":"v_minimummaximum_f32","records":1,"summary":"Select the IEEE minimum() of the first two single-precision float inputs and then select the IEEE maximum() of that result and third single-precision…","page":"https://instructionsets.com/amdgpu/v_minimummaximum_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_minimummaximum_f32.json"},{"mnemonic":"v_minmax_f16","slug":"v_minmax_f16","records":1,"summary":"Select the minimum of the first two half-precision float inputs and then select the maximum of that result and third half-precision float input.","page":"https://instructionsets.com/amdgpu/v_minmax_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_minmax_f16.json","aliases":["v_minmax_num_f16"]},{"mnemonic":"v_minmax_f32","slug":"v_minmax_f32","records":1,"summary":"Select the minimum of the first two single-precision float inputs and then select the maximum of that result and third single-precision float input.","page":"https://instructionsets.com/amdgpu/v_minmax_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_minmax_f32.json","aliases":["v_minmax_num_f32"]},{"mnemonic":"v_minmax_i32","slug":"v_minmax_i32","records":1,"summary":"Select the minimum of the first two signed 32-bit integer inputs and then select the maximum of that result and third signed 32-bit integer input.","page":"https://instructionsets.com/amdgpu/v_minmax_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_minmax_i32.json"},{"mnemonic":"v_minmax_num_f16","slug":"v_minmax_num_f16","records":1,"summary":"Select the IEEE minimumNumber() of the first two half-precision float inputs and then select the IEEE maximumNumber() of that result and third…","page":"https://instructionsets.com/amdgpu/v_minmax_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_minmax_num_f16.json","aliases":["v_minmax_f16"]},{"mnemonic":"v_minmax_num_f32","slug":"v_minmax_num_f32","records":1,"summary":"Select the IEEE minimumNumber() of the first two single-precision float inputs and then select the IEEE maximumNumber() of that result and third…","page":"https://instructionsets.com/amdgpu/v_minmax_num_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_minmax_num_f32.json","aliases":["v_minmax_f32"]},{"mnemonic":"v_minmax_u32","slug":"v_minmax_u32","records":1,"summary":"Select the minimum of the first two unsigned 32-bit integer inputs and then select the maximum of that result and third unsigned 32-bit integer input.","page":"https://instructionsets.com/amdgpu/v_minmax_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_minmax_u32.json"},{"mnemonic":"v_mov_b16","slug":"v_mov_b16","records":1,"summary":"Move 16-bit data from a vector input into a vector register.","page":"https://instructionsets.com/amdgpu/v_mov_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mov_b16.json"},{"mnemonic":"v_mov_b32","slug":"v_mov_b32","records":1,"summary":"Move 32-bit data from a vector input into a vector register.","page":"https://instructionsets.com/amdgpu/v_mov_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mov_b32.json"},{"mnemonic":"v_mov_b64","slug":"v_mov_b64","records":1,"summary":"Move data from a 64-bit vector input into a vector register.","page":"https://instructionsets.com/amdgpu/v_mov_b64/","api":"https://instructionsets.com/api/v1/amdgpu/v_mov_b64.json"},{"mnemonic":"v_movreld_b32","slug":"v_movreld_b32","records":1,"summary":"Move data from a vector input into a relatively-indexed vector register.","page":"https://instructionsets.com/amdgpu/v_movreld_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_movreld_b32.json"},{"mnemonic":"v_movrels_b32","slug":"v_movrels_b32","records":1,"summary":"Move data from a relatively-indexed vector register into another vector register.","page":"https://instructionsets.com/amdgpu/v_movrels_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_movrels_b32.json"},{"mnemonic":"v_movrelsd_2_b32","slug":"v_movrelsd_2_b32","records":1,"summary":"Move data from a relatively-indexed vector register into another relatively-indexed vector register, using different offsets for each index.","page":"https://instructionsets.com/amdgpu/v_movrelsd_2_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_movrelsd_2_b32.json"},{"mnemonic":"v_movrelsd_b32","slug":"v_movrelsd_b32","records":1,"summary":"Move data from a relatively-indexed vector register into another relatively-indexed vector register.","page":"https://instructionsets.com/amdgpu/v_movrelsd_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_movrelsd_b32.json"},{"mnemonic":"v_mqsad_pk_u16_u8","slug":"v_mqsad_pk_u16_u8","records":1,"summary":"Perform the V_MSAD_U8 operation four times using different slices of the first array, all entries of the second array and each entry of the third…","page":"https://instructionsets.com/amdgpu/v_mqsad_pk_u16_u8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mqsad_pk_u16_u8.json"},{"mnemonic":"v_mqsad_u32_u8","slug":"v_mqsad_u32_u8","records":1,"summary":"Perform the V_MSAD_U8 operation four times using different slices of the first array, all entries of the second array and each entry of the third…","page":"https://instructionsets.com/amdgpu/v_mqsad_u32_u8/","api":"https://instructionsets.com/api/v1/amdgpu/v_mqsad_u32_u8.json"},{"mnemonic":"v_msad_u8","slug":"v_msad_u8","records":1,"summary":"Calculate the sum of absolute differences of elements in two packed 4-component unsigned 8-bit integer inputs, except that elements where the second…","page":"https://instructionsets.com/amdgpu/v_msad_u8/","api":"https://instructionsets.com/api/v1/amdgpu/v_msad_u8.json"},{"mnemonic":"v_mul_f16","slug":"v_mul_f16","records":1,"summary":"Multiply two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_f16.json"},{"mnemonic":"v_mul_f32","slug":"v_mul_f32","records":1,"summary":"Per-lane single-precision floating-point multiply.","page":"https://instructionsets.com/amdgpu/v_mul_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_f32.json"},{"mnemonic":"v_mul_f64","slug":"v_mul_f64","records":1,"summary":"Multiply two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_f64.json"},{"mnemonic":"v_mul_f64_pseudo","slug":"v_mul_f64_pseudo","records":1,"summary":"AMDGPU VOP2 vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mul_f64_pseudo/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_f64_pseudo.json"},{"mnemonic":"v_mul_hi_i32","slug":"v_mul_hi_i32","records":1,"summary":"Multiply two signed 32-bit integer inputs and store the high 32 bits of the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_hi_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_hi_i32.json"},{"mnemonic":"v_mul_hi_i32_i24","slug":"v_mul_hi_i32_i24","records":1,"summary":"Multiply two signed 24-bit integer inputs and store the high 32 bits of the result as a signed 32-bit integer into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_hi_i32_i24/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_hi_i32_i24.json"},{"mnemonic":"v_mul_hi_u32","slug":"v_mul_hi_u32","records":1,"summary":"Multiply two unsigned 32-bit integer inputs and store the high 32 bits of the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_hi_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_hi_u32.json"},{"mnemonic":"v_mul_hi_u32_u24","slug":"v_mul_hi_u32_u24","records":1,"summary":"Multiply two unsigned 24-bit integer inputs and store the high 32 bits of the result as an unsigned 32-bit integer into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_hi_u32_u24/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_hi_u32_u24.json"},{"mnemonic":"v_mul_i32_i24","slug":"v_mul_i32_i24","records":1,"summary":"Multiply two signed 24-bit integer inputs and store the result as a signed 32-bit integer into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_i32_i24/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_i32_i24.json"},{"mnemonic":"v_mul_legacy_f32","slug":"v_mul_legacy_f32","records":1,"summary":"Multiply two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_legacy_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_legacy_f32.json","aliases":["v_mul_dx9_zero_f32"]},{"mnemonic":"v_mul_lo_i32","slug":"v_mul_lo_i32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mul_lo_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_lo_i32.json"},{"mnemonic":"v_mul_lo_u16","slug":"v_mul_lo_u16","records":1,"summary":"Multiply two unsigned 16-bit integer inputs and store the low bits of the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_lo_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_lo_u16.json"},{"mnemonic":"v_mul_lo_u32","slug":"v_mul_lo_u32","records":1,"summary":"Per-lane 32-bit unsigned multiply, low half of the product.","page":"https://instructionsets.com/amdgpu/v_mul_lo_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_lo_u32.json"},{"mnemonic":"v_mul_u32_u24","slug":"v_mul_u32_u24","records":1,"summary":"Multiply two unsigned 24-bit integer inputs and store the result as an unsigned 32-bit integer into a vector register.","page":"https://instructionsets.com/amdgpu/v_mul_u32_u24/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_u32_u24.json"},{"mnemonic":"v_mul_u64","slug":"v_mul_u64","records":1,"summary":"AMDGPU VOP2 vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_mul_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_mul_u64.json"},{"mnemonic":"v_mullit_f32","slug":"v_mullit_f32","records":1,"summary":"Multiply two floating point inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_mullit_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_mullit_f32.json"},{"mnemonic":"v_nop","slug":"v_nop","records":1,"summary":"Do nothing.","page":"https://instructionsets.com/amdgpu/v_nop/","api":"https://instructionsets.com/api/v1/amdgpu/v_nop.json"},{"mnemonic":"v_not_b16","slug":"v_not_b16","records":1,"summary":"Calculate bitwise negation on a vector input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_not_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_not_b16.json"},{"mnemonic":"v_not_b32","slug":"v_not_b32","records":1,"summary":"Calculate bitwise negation on a vector input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_not_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_not_b32.json"},{"mnemonic":"v_or3_b32","slug":"v_or3_b32","records":1,"summary":"Calculate the bitwise OR of three vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_or3_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_or3_b32.json"},{"mnemonic":"v_or_b16","slug":"v_or_b16","records":1,"summary":"Calculate bitwise OR on two vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_or_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_or_b16.json"},{"mnemonic":"v_or_b16_fake16","slug":"v_or_b16_fake16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_or_b16_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_or_b16_fake16.json"},{"mnemonic":"v_or_b16_t16","slug":"v_or_b16_t16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_or_b16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_or_b16_t16.json"},{"mnemonic":"v_or_b32","slug":"v_or_b32","records":1,"summary":"Calculate bitwise OR on two vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_or_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_or_b32.json"},{"mnemonic":"v_pack_b32_f16","slug":"v_pack_b32_f16","records":1,"summary":"Pack two half-precision float values into a single 32-bit value and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_pack_b32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pack_b32_f16.json"},{"mnemonic":"v_perm_b32","slug":"v_perm_b32","records":1,"summary":"Permute a 64-bit value constructed from two vector inputs (most significant bits come from the first input) using a per-lane selector from the third…","page":"https://instructionsets.com/amdgpu/v_perm_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_perm_b32.json"},{"mnemonic":"v_perm_pk16_b4_u4","slug":"v_perm_pk16_b4_u4","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_perm_pk16_b4_u4/","api":"https://instructionsets.com/api/v1/amdgpu/v_perm_pk16_b4_u4.json"},{"mnemonic":"v_perm_pk16_b6_u4","slug":"v_perm_pk16_b6_u4","records":1,"summary":"AMDGPU VOP3 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_perm_pk16_b6_u4/","api":"https://instructionsets.com/api/v1/amdgpu/v_perm_pk16_b6_u4.json"},{"mnemonic":"v_perm_pk16_b8_u4","slug":"v_perm_pk16_b8_u4","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b8 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_perm_pk16_b8_u4/","api":"https://instructionsets.com/api/v1/amdgpu/v_perm_pk16_b8_u4.json"},{"mnemonic":"v_permlane16_b32","slug":"v_permlane16_b32","records":1,"summary":"Perform arbitrary gather-style operation within a row (16 contiguous lanes).","page":"https://instructionsets.com/amdgpu/v_permlane16_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane16_b32.json"},{"mnemonic":"v_permlane16_swap_b32","slug":"v_permlane16_swap_b32","records":1,"summary":"Swap data between two vector registers. Odd rows of the first operand are swapped with even rows of the second operand (one row is 16 lanes).","page":"https://instructionsets.com/amdgpu/v_permlane16_swap_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane16_swap_b32.json"},{"mnemonic":"v_permlane16_var_b32","slug":"v_permlane16_var_b32","records":1,"summary":"Perform arbitrary gather-style operation within a row (16 contiguous lanes).","page":"https://instructionsets.com/amdgpu/v_permlane16_var_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane16_var_b32.json"},{"mnemonic":"v_permlane32_swap_b32","slug":"v_permlane32_swap_b32","records":1,"summary":"Swap data between two vector registers. Rows 2 and 3 of the first operand are swapped with rows 0 and 1 of the second operand (one row is 16 lanes).","page":"https://instructionsets.com/amdgpu/v_permlane32_swap_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane32_swap_b32.json"},{"mnemonic":"v_permlane64_b32","slug":"v_permlane64_b32","records":1,"summary":"Perform a specific permutation across lanes where the high half and low half of a wave64 are swapped. Performs no operation in wave32 mode.","page":"https://instructionsets.com/amdgpu/v_permlane64_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane64_b32.json"},{"mnemonic":"v_permlane_bcast_b32","slug":"v_permlane_bcast_b32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_permlane_bcast_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane_bcast_b32.json"},{"mnemonic":"v_permlane_down_b32","slug":"v_permlane_down_b32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_permlane_down_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane_down_b32.json"},{"mnemonic":"v_permlane_idx_gen_b32","slug":"v_permlane_idx_gen_b32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_permlane_idx_gen_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane_idx_gen_b32.json"},{"mnemonic":"v_permlane_up_b32","slug":"v_permlane_up_b32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_permlane_up_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane_up_b32.json"},{"mnemonic":"v_permlane_xor_b32","slug":"v_permlane_xor_b32","records":1,"summary":"AMDGPU VOP3 vector instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_permlane_xor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlane_xor_b32.json"},{"mnemonic":"v_permlanex16_b32","slug":"v_permlanex16_b32","records":1,"summary":"Perform arbitrary gather-style operation across two rows (each row is 16 contiguous lanes).","page":"https://instructionsets.com/amdgpu/v_permlanex16_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlanex16_b32.json"},{"mnemonic":"v_permlanex16_var_b32","slug":"v_permlanex16_var_b32","records":1,"summary":"Perform arbitrary gather-style operation across two rows (each row is 16 contiguous lanes).","page":"https://instructionsets.com/amdgpu/v_permlanex16_var_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_permlanex16_var_b32.json"},{"mnemonic":"v_pipeflush","slug":"v_pipeflush","records":1,"summary":"Flush the vector ALU pipeline through the destination cache.","page":"https://instructionsets.com/amdgpu/v_pipeflush/","api":"https://instructionsets.com/api/v1/amdgpu/v_pipeflush.json"},{"mnemonic":"v_pk_add_bf16","slug":"v_pk_add_bf16","records":1,"summary":"AMDGPU VOP3P vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_add_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_bf16.json"},{"mnemonic":"v_pk_add_f16","slug":"v_pk_add_f16","records":1,"summary":"Add two packed half-precision float inputs component-wise and store the result into a vector register. No carry- in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_pk_add_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_f16.json"},{"mnemonic":"v_pk_add_f32","slug":"v_pk_add_f32","records":1,"summary":"Add two packed single-precision float inputs component-wise and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_pk_add_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_f32.json"},{"mnemonic":"v_pk_add_f64","slug":"v_pk_add_f64","records":1,"summary":"AMDGPU VOP3P vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_add_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_f64.json"},{"mnemonic":"v_pk_add_i16","slug":"v_pk_add_i16","records":1,"summary":"Add two packed signed 16-bit integer inputs component-wise and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_pk_add_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_i16.json"},{"mnemonic":"v_pk_add_max_i16","slug":"v_pk_add_max_i16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on i16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_add_max_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_max_i16.json"},{"mnemonic":"v_pk_add_max_u16","slug":"v_pk_add_max_u16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on u16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_add_max_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_max_u16.json"},{"mnemonic":"v_pk_add_min_i16","slug":"v_pk_add_min_i16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on i16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_add_min_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_min_i16.json"},{"mnemonic":"v_pk_add_min_u16","slug":"v_pk_add_min_u16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on u16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_add_min_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_min_u16.json"},{"mnemonic":"v_pk_add_nc_u64","slug":"v_pk_add_nc_u64","records":1,"summary":"AMDGPU VOP3P vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_add_nc_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_nc_u64.json"},{"mnemonic":"v_pk_add_u16","slug":"v_pk_add_u16","records":1,"summary":"Add two packed unsigned 16-bit integer inputs component-wise and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_pk_add_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_add_u16.json"},{"mnemonic":"v_pk_ashrrev_i16","slug":"v_pk_ashrrev_i16","records":1,"summary":"Given a packed shift count in the first vector input, calculate the component-wise arithmetic shift right (preserving sign bit) of the second packed…","page":"https://instructionsets.com/amdgpu/v_pk_ashrrev_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_ashrrev_i16.json"},{"mnemonic":"v_pk_fma_bf16","slug":"v_pk_fma_bf16","records":1,"summary":"AMDGPU VOP3P vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_fma_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_fma_bf16.json"},{"mnemonic":"v_pk_fma_f16","slug":"v_pk_fma_f16","records":1,"summary":"Multiply two packed half-precision float inputs component-wise and add a third input component-wise using fused multiply add, and store the result…","page":"https://instructionsets.com/amdgpu/v_pk_fma_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_fma_f16.json"},{"mnemonic":"v_pk_fma_f32","slug":"v_pk_fma_f32","records":1,"summary":"Multiply two packed single-precision float inputs component-wise and add a third input component-wise using fused multiply add, and store the result…","page":"https://instructionsets.com/amdgpu/v_pk_fma_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_fma_f32.json"},{"mnemonic":"v_pk_fma_f64","slug":"v_pk_fma_f64","records":1,"summary":"AMDGPU VOP3P vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_fma_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_fma_f64.json"},{"mnemonic":"v_pk_fmac_f16","slug":"v_pk_fmac_f16","records":1,"summary":"Multiply two packed half-precision float inputs component-wise and accumulate the result into the destination register using fused multiply add.","page":"https://instructionsets.com/amdgpu/v_pk_fmac_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_fmac_f16.json"},{"mnemonic":"v_pk_lshl_add_u64","slug":"v_pk_lshl_add_u64","records":1,"summary":"AMDGPU VOP3P vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_lshl_add_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_lshl_add_u64.json"},{"mnemonic":"v_pk_lshlrev_b16","slug":"v_pk_lshlrev_b16","records":1,"summary":"Given a packed shift count in the first vector input, calculate the component-wise logical shift left of the second packed vector input and store the…","page":"https://instructionsets.com/amdgpu/v_pk_lshlrev_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_lshlrev_b16.json"},{"mnemonic":"v_pk_lshrrev_b16","slug":"v_pk_lshrrev_b16","records":1,"summary":"Given a packed shift count in the first vector input, calculate the component-wise logical shift right of the second packed vector input and store…","page":"https://instructionsets.com/amdgpu/v_pk_lshrrev_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_lshrrev_b16.json"},{"mnemonic":"v_pk_mad_i16","slug":"v_pk_mad_i16","records":1,"summary":"Multiply two packed signed 16-bit integer inputs component-wise, add a packed signed 16-bit integer value from a third input component-wise, and…","page":"https://instructionsets.com/amdgpu/v_pk_mad_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_mad_i16.json"},{"mnemonic":"v_pk_mad_u16","slug":"v_pk_mad_u16","records":1,"summary":"Multiply two packed unsigned 16-bit integer inputs component-wise, add a packed unsigned 16-bit integer value from a third input component-wise, and…","page":"https://instructionsets.com/amdgpu/v_pk_mad_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_mad_u16.json"},{"mnemonic":"v_pk_max3_i16","slug":"v_pk_max3_i16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on i16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_max3_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_max3_i16.json"},{"mnemonic":"v_pk_max3_num_f16","slug":"v_pk_max3_num_f16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_max3_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_max3_num_f16.json"},{"mnemonic":"v_pk_max3_u16","slug":"v_pk_max3_u16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on u16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_max3_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_max3_u16.json"},{"mnemonic":"v_pk_max_f16","slug":"v_pk_max_f16","records":1,"summary":"Select the component-wise maximum of two packed half-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_max_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_max_f16.json","aliases":["v_pk_max_num_f16"]},{"mnemonic":"v_pk_max_i16","slug":"v_pk_max_i16","records":1,"summary":"Select the component-wise maximum of two packed signed 16-bit integer inputs and store the selected values into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_max_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_max_i16.json"},{"mnemonic":"v_pk_max_num_bf16","slug":"v_pk_max_num_bf16","records":1,"summary":"AMDGPU VOP3P vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_max_num_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_max_num_bf16.json"},{"mnemonic":"v_pk_max_num_f16","slug":"v_pk_max_num_f16","records":1,"summary":"Select the component-wise IEEE maximumNumber() of two packed half-precision float inputs and store the selected values into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_max_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_max_num_f16.json","aliases":["v_pk_max_f16"]},{"mnemonic":"v_pk_max_num_f64","slug":"v_pk_max_num_f64","records":1,"summary":"AMDGPU VOP3P vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_max_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_max_num_f64.json"},{"mnemonic":"v_pk_max_u16","slug":"v_pk_max_u16","records":1,"summary":"Select the component-wise maximum of two packed unsigned 16-bit integer inputs and store the selected values into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_max_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_max_u16.json"},{"mnemonic":"v_pk_maximum3_f16","slug":"v_pk_maximum3_f16","records":1,"summary":"Select the component-wise IEEE maximum() of three half-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_maximum3_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_maximum3_f16.json"},{"mnemonic":"v_pk_maximum_f16","slug":"v_pk_maximum_f16","records":1,"summary":"Select the component-wise IEEE maximum() of two packed half-precision float inputs and store the selected values into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_maximum_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_maximum_f16.json"},{"mnemonic":"v_pk_min3_i16","slug":"v_pk_min3_i16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on i16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_min3_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_min3_i16.json"},{"mnemonic":"v_pk_min3_num_f16","slug":"v_pk_min3_num_f16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_min3_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_min3_num_f16.json"},{"mnemonic":"v_pk_min3_u16","slug":"v_pk_min3_u16","records":1,"summary":"AMDGPU VOP3P vector instruction operating on u16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_min3_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_min3_u16.json"},{"mnemonic":"v_pk_min_f16","slug":"v_pk_min_f16","records":1,"summary":"Select the component-wise minimum of two packed half-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_min_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_min_f16.json","aliases":["v_pk_min_num_f16"]},{"mnemonic":"v_pk_min_i16","slug":"v_pk_min_i16","records":1,"summary":"Select the component-wise minimum of two packed signed 16-bit integer inputs and store the selected values into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_min_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_min_i16.json"},{"mnemonic":"v_pk_min_num_bf16","slug":"v_pk_min_num_bf16","records":1,"summary":"AMDGPU VOP3P vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_min_num_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_min_num_bf16.json"},{"mnemonic":"v_pk_min_num_f16","slug":"v_pk_min_num_f16","records":1,"summary":"Select the component-wise IEEE minimumNumber() of two packed half-precision float inputs and store the selected values into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_min_num_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_min_num_f16.json","aliases":["v_pk_min_f16"]},{"mnemonic":"v_pk_min_num_f64","slug":"v_pk_min_num_f64","records":1,"summary":"AMDGPU VOP3P vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_min_num_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_min_num_f64.json"},{"mnemonic":"v_pk_min_u16","slug":"v_pk_min_u16","records":1,"summary":"Select the component-wise minimum of two packed unsigned 16-bit integer inputs and store the selected values into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_min_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_min_u16.json"},{"mnemonic":"v_pk_minimum3_f16","slug":"v_pk_minimum3_f16","records":1,"summary":"Select the component-wise IEEE minimum() of three half-precision float inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_minimum3_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_minimum3_f16.json"},{"mnemonic":"v_pk_minimum_f16","slug":"v_pk_minimum_f16","records":1,"summary":"Select the component-wise IEEE minimum() of two packed half-precision float inputs and store the selected values into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_minimum_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_minimum_f16.json"},{"mnemonic":"v_pk_mov_b32","slug":"v_pk_mov_b32","records":1,"summary":"Move data from two vector inputs into two vector registers.","page":"https://instructionsets.com/amdgpu/v_pk_mov_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_mov_b32.json"},{"mnemonic":"v_pk_mul_bf16","slug":"v_pk_mul_bf16","records":1,"summary":"AMDGPU VOP3P vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_mul_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_mul_bf16.json"},{"mnemonic":"v_pk_mul_f16","slug":"v_pk_mul_f16","records":1,"summary":"Multiply two packed half-precision float inputs component-wise and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_mul_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_mul_f16.json"},{"mnemonic":"v_pk_mul_f32","slug":"v_pk_mul_f32","records":1,"summary":"Multiply two packed single-precision float inputs component-wise and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_mul_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_mul_f32.json"},{"mnemonic":"v_pk_mul_f64","slug":"v_pk_mul_f64","records":1,"summary":"AMDGPU VOP3P vector instruction operating on f64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_mul_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_mul_f64.json"},{"mnemonic":"v_pk_mul_lo_u16","slug":"v_pk_mul_lo_u16","records":1,"summary":"Multiply two packed unsigned 16-bit integer inputs component-wise and store the low bits of each resulting component into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_mul_lo_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_mul_lo_u16.json"},{"mnemonic":"v_pk_sub_i16","slug":"v_pk_sub_i16","records":1,"summary":"Subtract the second packed signed 16-bit integer input from the first input component-wise and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_sub_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_sub_i16.json"},{"mnemonic":"v_pk_sub_nc_u64","slug":"v_pk_sub_nc_u64","records":1,"summary":"AMDGPU VOP3P vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_pk_sub_nc_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_sub_nc_u64.json"},{"mnemonic":"v_pk_sub_u16","slug":"v_pk_sub_u16","records":1,"summary":"Subtract the second packed unsigned 16-bit integer input from the first input component-wise and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_pk_sub_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_pk_sub_u16.json"},{"mnemonic":"v_prng_b32","slug":"v_prng_b32","records":1,"summary":"Generate a pseudorandom number using an LFSR (linear feedback shift register) seeded with the vector input, then store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_prng_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_prng_b32.json"},{"mnemonic":"v_qsad_pk_u16_u8","slug":"v_qsad_pk_u16_u8","records":1,"summary":"Perform the V_SAD_U8 operation four times using different slices of the first array, all entries of the second array and each entry of the third…","page":"https://instructionsets.com/amdgpu/v_qsad_pk_u16_u8/","api":"https://instructionsets.com/api/v1/amdgpu/v_qsad_pk_u16_u8.json"},{"mnemonic":"v_rcp_bf16","slug":"v_rcp_bf16","records":1,"summary":"AMDGPU VOP1 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_rcp_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_rcp_bf16.json"},{"mnemonic":"v_rcp_f16","slug":"v_rcp_f16","records":1,"summary":"Calculate the reciprocal of the half-precision float input using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_rcp_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_rcp_f16.json"},{"mnemonic":"v_rcp_f32","slug":"v_rcp_f32","records":1,"summary":"Calculate the reciprocal of the single-precision float input using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_rcp_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_rcp_f32.json"},{"mnemonic":"v_rcp_f64","slug":"v_rcp_f64","records":1,"summary":"Calculate the reciprocal of the double-precision float input using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_rcp_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_rcp_f64.json"},{"mnemonic":"v_rcp_iflag_f32","slug":"v_rcp_iflag_f32","records":1,"summary":"Calculate the reciprocal of the vector float input in a manner suitable for integer division and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_rcp_iflag_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_rcp_iflag_f32.json"},{"mnemonic":"v_readfirstlane_b32","slug":"v_readfirstlane_b32","records":1,"summary":"Read the value of a VGPR from the first active lane into a scalar register.","page":"https://instructionsets.com/amdgpu/v_readfirstlane_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_readfirstlane_b32.json"},{"mnemonic":"v_readlane_b32","slug":"v_readlane_b32","records":1,"summary":"Read the scalar value in the specified lane of the first input where the lane select is in the second input. Store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_readlane_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_readlane_b32.json"},{"mnemonic":"v_rndne_f16","slug":"v_rndne_f16","records":1,"summary":"Round the half-precision float input to the nearest even integer and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_rndne_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_rndne_f16.json"},{"mnemonic":"v_rndne_f32","slug":"v_rndne_f32","records":1,"summary":"Round the single-precision float input to the nearest even integer and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_rndne_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_rndne_f32.json"},{"mnemonic":"v_rndne_f64","slug":"v_rndne_f64","records":1,"summary":"Round the double-precision float input to the nearest even integer and store the result in floating point format into a vector register.","page":"https://instructionsets.com/amdgpu/v_rndne_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_rndne_f64.json"},{"mnemonic":"v_rsq_bf16","slug":"v_rsq_bf16","records":1,"summary":"AMDGPU VOP1 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_rsq_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_rsq_bf16.json"},{"mnemonic":"v_rsq_f16","slug":"v_rsq_f16","records":1,"summary":"Calculate the reciprocal of the square root of the half-precision float input using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_rsq_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_rsq_f16.json"},{"mnemonic":"v_rsq_f32","slug":"v_rsq_f32","records":1,"summary":"Per-lane fast approximate reciprocal square root.","page":"https://instructionsets.com/amdgpu/v_rsq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_rsq_f32.json"},{"mnemonic":"v_rsq_f64","slug":"v_rsq_f64","records":1,"summary":"Calculate the reciprocal of the square root of the double-precision float input using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_rsq_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_rsq_f64.json"},{"mnemonic":"v_s_exp_f16","slug":"v_s_exp_f16","records":1,"summary":"Calculate 2 raised to the power of the half-precision float input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_exp_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_exp_f16.json"},{"mnemonic":"v_s_exp_f32","slug":"v_s_exp_f32","records":1,"summary":"Calculate 2 raised to the power of the single-precision float input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_exp_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_exp_f32.json"},{"mnemonic":"v_s_log_f16","slug":"v_s_log_f16","records":1,"summary":"Calculate the base 2 logarithm of the half-precision float input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_log_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_log_f16.json"},{"mnemonic":"v_s_log_f32","slug":"v_s_log_f32","records":1,"summary":"Calculate the base 2 logarithm of the single-precision float input and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_log_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_log_f32.json"},{"mnemonic":"v_s_rcp_f16","slug":"v_s_rcp_f16","records":1,"summary":"Calculate the reciprocal of the half-precision float input using IEEE rules and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_rcp_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_rcp_f16.json"},{"mnemonic":"v_s_rcp_f32","slug":"v_s_rcp_f32","records":1,"summary":"Calculate the reciprocal of the single-precision float input using IEEE rules and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_rcp_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_rcp_f32.json"},{"mnemonic":"v_s_rsq_f16","slug":"v_s_rsq_f16","records":1,"summary":"Calculate the reciprocal of the square root of the half-precision float input using IEEE rules and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_rsq_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_rsq_f16.json"},{"mnemonic":"v_s_rsq_f32","slug":"v_s_rsq_f32","records":1,"summary":"Calculate the reciprocal of the square root of the single-precision float input using IEEE rules and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_rsq_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_rsq_f32.json"},{"mnemonic":"v_s_sqrt_f16","slug":"v_s_sqrt_f16","records":1,"summary":"Calculate the square root of the half-precision float input using IEEE rules and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_sqrt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_sqrt_f16.json"},{"mnemonic":"v_s_sqrt_f32","slug":"v_s_sqrt_f32","records":1,"summary":"Calculate the square root of the single-precision float input using IEEE rules and store the result into a scalar register.","page":"https://instructionsets.com/amdgpu/v_s_sqrt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_s_sqrt_f32.json"},{"mnemonic":"v_sad_hi_u8","slug":"v_sad_hi_u8","records":1,"summary":"Calculate the sum of absolute differences of elements in two packed 4-component unsigned 8-bit integer inputs, shift the sum left by 16 bits, add an…","page":"https://instructionsets.com/amdgpu/v_sad_hi_u8/","api":"https://instructionsets.com/api/v1/amdgpu/v_sad_hi_u8.json"},{"mnemonic":"v_sad_u16","slug":"v_sad_u16","records":1,"summary":"Calculate the sum of absolute differences of elements in two packed 2-component unsigned 16-bit integer inputs, add an unsigned 32-bit integer value…","page":"https://instructionsets.com/amdgpu/v_sad_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sad_u16.json"},{"mnemonic":"v_sad_u32","slug":"v_sad_u32","records":1,"summary":"Calculate the absolute difference of two unsigned 32-bit integer inputs, add an unsigned 32-bit integer value from the third input and store the…","page":"https://instructionsets.com/amdgpu/v_sad_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sad_u32.json"},{"mnemonic":"v_sad_u8","slug":"v_sad_u8","records":1,"summary":"Calculate the sum of absolute differences of elements in two packed 4-component unsigned 8-bit integer inputs, add an unsigned 32-bit integer value…","page":"https://instructionsets.com/amdgpu/v_sad_u8/","api":"https://instructionsets.com/api/v1/amdgpu/v_sad_u8.json"},{"mnemonic":"v_sat_pk4_i4_i8","slug":"v_sat_pk4_i4_i8","records":1,"summary":"AMDGPU VOP1 vector instruction operating on i8 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_sat_pk4_i4_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_sat_pk4_i4_i8.json"},{"mnemonic":"v_sat_pk4_u4_u8","slug":"v_sat_pk4_u4_u8","records":1,"summary":"AMDGPU VOP1 vector instruction operating on u8 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_sat_pk4_u4_u8/","api":"https://instructionsets.com/api/v1/amdgpu/v_sat_pk4_u4_u8.json"},{"mnemonic":"v_sat_pk_u8_i16","slug":"v_sat_pk_u8_i16","records":1,"summary":"Given 2 signed 16-bit integer inputs, saturate each input over an unsigned 8-bit integer range, pack the resulting values into a packed 16-bit value…","page":"https://instructionsets.com/amdgpu/v_sat_pk_u8_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sat_pk_u8_i16.json"},{"mnemonic":"v_screen_partition_4se_b32","slug":"v_screen_partition_4se_b32","records":1,"summary":"4SE version of LUT instruction for screen partitioning/filtering.","page":"https://instructionsets.com/amdgpu/v_screen_partition_4se_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_screen_partition_4se_b32.json"},{"mnemonic":"v_sin_bf16","slug":"v_sin_bf16","records":1,"summary":"AMDGPU VOP1 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_sin_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sin_bf16.json"},{"mnemonic":"v_sin_f16","slug":"v_sin_f16","records":1,"summary":"Calculate the trigonometric sine of a half-precision float value using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_sin_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sin_f16.json"},{"mnemonic":"v_sin_f32","slug":"v_sin_f32","records":1,"summary":"Per-lane fast approximate sine.","page":"https://instructionsets.com/amdgpu/v_sin_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sin_f32.json"},{"mnemonic":"v_smfmac_f32_16x16x128_bf8_bf8","slug":"v_smfmac_f32_16x16x128_bf8_bf8","records":1,"summary":"Multiply the 16x128 sparse matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix stored…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x128_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x128_bf8_bf8.json"},{"mnemonic":"v_smfmac_f32_16x16x128_bf8_fp8","slug":"v_smfmac_f32_16x16x128_bf8_fp8","records":1,"summary":"Multiply the 16x128 sparse matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix stored…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x128_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x128_bf8_fp8.json"},{"mnemonic":"v_smfmac_f32_16x16x128_fp8_bf8","slug":"v_smfmac_f32_16x16x128_fp8_bf8","records":1,"summary":"Multiply the 16x128 sparse matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix stored…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x128_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x128_fp8_bf8.json"},{"mnemonic":"v_smfmac_f32_16x16x128_fp8_fp8","slug":"v_smfmac_f32_16x16x128_fp8_fp8","records":1,"summary":"Multiply the 16x128 sparse matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix stored…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x128_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x128_fp8_fp8.json"},{"mnemonic":"v_smfmac_f32_16x16x128bf8bf8","slug":"v_smfmac_f32_16x16x128bf8bf8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x128bf8bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x128bf8bf8.json"},{"mnemonic":"v_smfmac_f32_16x16x128bf8fp8","slug":"v_smfmac_f32_16x16x128bf8fp8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x128bf8fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x128bf8fp8.json"},{"mnemonic":"v_smfmac_f32_16x16x128fp8bf8","slug":"v_smfmac_f32_16x16x128fp8bf8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x128fp8bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x128fp8bf8.json"},{"mnemonic":"v_smfmac_f32_16x16x128fp8fp8","slug":"v_smfmac_f32_16x16x128fp8fp8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x128fp8fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x128fp8fp8.json"},{"mnemonic":"v_smfmac_f32_16x16x32_bf16","slug":"v_smfmac_f32_16x16x32_bf16","records":1,"summary":"Multiply the 16x32 sparse matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x32_bf16.json","aliases":["v_smfmac_f32_16x16x32bf16"]},{"mnemonic":"v_smfmac_f32_16x16x32_f16","slug":"v_smfmac_f32_16x16x32_f16","records":1,"summary":"Multiply the 16x32 sparse matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x32_f16.json","aliases":["v_smfmac_f32_16x16x32f16"]},{"mnemonic":"v_smfmac_f32_16x16x32bf16","slug":"v_smfmac_f32_16x16x32bf16","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x32bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x32bf16.json","aliases":["v_smfmac_f32_16x16x32_bf16"]},{"mnemonic":"v_smfmac_f32_16x16x32f16","slug":"v_smfmac_f32_16x16x32f16","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x32f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x32f16.json","aliases":["v_smfmac_f32_16x16x32_f16"]},{"mnemonic":"v_smfmac_f32_16x16x64_bf16","slug":"v_smfmac_f32_16x16x64_bf16","records":1,"summary":"Multiply the 16x64 sparse matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64_bf16.json"},{"mnemonic":"v_smfmac_f32_16x16x64_bf8_bf8","slug":"v_smfmac_f32_16x16x64_bf8_bf8","records":1,"summary":"Multiply the 16x64 sparse matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64_bf8_bf8.json","aliases":["v_smfmac_f32_16x16x64bf8bf8"]},{"mnemonic":"v_smfmac_f32_16x16x64_bf8_fp8","slug":"v_smfmac_f32_16x16x64_bf8_fp8","records":1,"summary":"Multiply the 16x64 sparse matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64_bf8_fp8.json","aliases":["v_smfmac_f32_16x16x64bf8fp8"]},{"mnemonic":"v_smfmac_f32_16x16x64_f16","slug":"v_smfmac_f32_16x16x64_f16","records":1,"summary":"Multiply the 16x64 sparse matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64_f16.json"},{"mnemonic":"v_smfmac_f32_16x16x64_fp8_bf8","slug":"v_smfmac_f32_16x16x64_fp8_bf8","records":1,"summary":"Multiply the 16x64 sparse matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64_fp8_bf8.json","aliases":["v_smfmac_f32_16x16x64fp8bf8"]},{"mnemonic":"v_smfmac_f32_16x16x64_fp8_fp8","slug":"v_smfmac_f32_16x16x64_fp8_fp8","records":1,"summary":"Multiply the 16x64 sparse matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64_fp8_fp8.json","aliases":["v_smfmac_f32_16x16x64fp8fp8"]},{"mnemonic":"v_smfmac_f32_16x16x64bf16","slug":"v_smfmac_f32_16x16x64bf16","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64bf16.json"},{"mnemonic":"v_smfmac_f32_16x16x64bf8bf8","slug":"v_smfmac_f32_16x16x64bf8bf8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64bf8bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64bf8bf8.json","aliases":["v_smfmac_f32_16x16x64_bf8_bf8"]},{"mnemonic":"v_smfmac_f32_16x16x64bf8fp8","slug":"v_smfmac_f32_16x16x64bf8fp8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64bf8fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64bf8fp8.json","aliases":["v_smfmac_f32_16x16x64_bf8_fp8"]},{"mnemonic":"v_smfmac_f32_16x16x64f16","slug":"v_smfmac_f32_16x16x64f16","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64f16.json"},{"mnemonic":"v_smfmac_f32_16x16x64fp8bf8","slug":"v_smfmac_f32_16x16x64fp8bf8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64fp8bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64fp8bf8.json","aliases":["v_smfmac_f32_16x16x64_fp8_bf8"]},{"mnemonic":"v_smfmac_f32_16x16x64fp8fp8","slug":"v_smfmac_f32_16x16x64fp8fp8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_16x16x64fp8fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_16x16x64fp8fp8.json","aliases":["v_smfmac_f32_16x16x64_fp8_fp8"]},{"mnemonic":"v_smfmac_f32_32x32x16_bf16","slug":"v_smfmac_f32_32x32x16_bf16","records":1,"summary":"Multiply the 32x16 sparse matrix in the first input by the 16x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x16_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x16_bf16.json","aliases":["v_smfmac_f32_32x32x16bf16"]},{"mnemonic":"v_smfmac_f32_32x32x16_f16","slug":"v_smfmac_f32_32x32x16_f16","records":1,"summary":"Multiply the 32x16 sparse matrix in the first input by the 16x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x16_f16.json","aliases":["v_smfmac_f32_32x32x16f16"]},{"mnemonic":"v_smfmac_f32_32x32x16bf16","slug":"v_smfmac_f32_32x32x16bf16","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x16bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x16bf16.json","aliases":["v_smfmac_f32_32x32x16_bf16"]},{"mnemonic":"v_smfmac_f32_32x32x16f16","slug":"v_smfmac_f32_32x32x16f16","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x16f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x16f16.json","aliases":["v_smfmac_f32_32x32x16_f16"]},{"mnemonic":"v_smfmac_f32_32x32x32_bf16","slug":"v_smfmac_f32_32x32x32_bf16","records":1,"summary":"Multiply the 32x32 sparse matrix in the first input by the 32x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32_bf16.json"},{"mnemonic":"v_smfmac_f32_32x32x32_bf8_bf8","slug":"v_smfmac_f32_32x32x32_bf8_bf8","records":1,"summary":"Multiply the 32x32 sparse matrix in the first input by the 32x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32_bf8_bf8.json","aliases":["v_smfmac_f32_32x32x32bf8bf8"]},{"mnemonic":"v_smfmac_f32_32x32x32_bf8_fp8","slug":"v_smfmac_f32_32x32x32_bf8_fp8","records":1,"summary":"Multiply the 32x32 sparse matrix in the first input by the 32x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32_bf8_fp8.json","aliases":["v_smfmac_f32_32x32x32bf8fp8"]},{"mnemonic":"v_smfmac_f32_32x32x32_f16","slug":"v_smfmac_f32_32x32x32_f16","records":1,"summary":"Multiply the 32x32 sparse matrix in the first input by the 32x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32_f16.json"},{"mnemonic":"v_smfmac_f32_32x32x32_fp8_bf8","slug":"v_smfmac_f32_32x32x32_fp8_bf8","records":1,"summary":"Multiply the 32x32 sparse matrix in the first input by the 32x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32_fp8_bf8.json","aliases":["v_smfmac_f32_32x32x32fp8bf8"]},{"mnemonic":"v_smfmac_f32_32x32x32_fp8_fp8","slug":"v_smfmac_f32_32x32x32_fp8_fp8","records":1,"summary":"Multiply the 32x32 sparse matrix in the first input by the 32x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32_fp8_fp8.json","aliases":["v_smfmac_f32_32x32x32fp8fp8"]},{"mnemonic":"v_smfmac_f32_32x32x32bf16","slug":"v_smfmac_f32_32x32x32bf16","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32bf16.json"},{"mnemonic":"v_smfmac_f32_32x32x32bf8bf8","slug":"v_smfmac_f32_32x32x32bf8bf8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32bf8bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32bf8bf8.json","aliases":["v_smfmac_f32_32x32x32_bf8_bf8"]},{"mnemonic":"v_smfmac_f32_32x32x32bf8fp8","slug":"v_smfmac_f32_32x32x32bf8fp8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32bf8fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32bf8fp8.json","aliases":["v_smfmac_f32_32x32x32_bf8_fp8"]},{"mnemonic":"v_smfmac_f32_32x32x32f16","slug":"v_smfmac_f32_32x32x32f16","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32f16.json"},{"mnemonic":"v_smfmac_f32_32x32x32fp8bf8","slug":"v_smfmac_f32_32x32x32fp8bf8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32fp8bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32fp8bf8.json","aliases":["v_smfmac_f32_32x32x32_fp8_bf8"]},{"mnemonic":"v_smfmac_f32_32x32x32fp8fp8","slug":"v_smfmac_f32_32x32x32fp8fp8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x32fp8fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x32fp8fp8.json","aliases":["v_smfmac_f32_32x32x32_fp8_fp8"]},{"mnemonic":"v_smfmac_f32_32x32x64_bf8_bf8","slug":"v_smfmac_f32_32x32x64_bf8_bf8","records":1,"summary":"Multiply the 32x64 sparse matrix in the first input by the 64x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x64_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x64_bf8_bf8.json"},{"mnemonic":"v_smfmac_f32_32x32x64_bf8_fp8","slug":"v_smfmac_f32_32x32x64_bf8_fp8","records":1,"summary":"Multiply the 32x64 sparse matrix in the first input by the 64x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x64_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x64_bf8_fp8.json"},{"mnemonic":"v_smfmac_f32_32x32x64_fp8_bf8","slug":"v_smfmac_f32_32x32x64_fp8_bf8","records":1,"summary":"Multiply the 32x64 sparse matrix in the first input by the 64x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x64_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x64_fp8_bf8.json"},{"mnemonic":"v_smfmac_f32_32x32x64_fp8_fp8","slug":"v_smfmac_f32_32x32x64_fp8_fp8","records":1,"summary":"Multiply the 32x64 sparse matrix in the first input by the 64x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x64_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x64_fp8_fp8.json"},{"mnemonic":"v_smfmac_f32_32x32x64bf8bf8","slug":"v_smfmac_f32_32x32x64bf8bf8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x64bf8bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x64bf8bf8.json"},{"mnemonic":"v_smfmac_f32_32x32x64bf8fp8","slug":"v_smfmac_f32_32x32x64bf8fp8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x64bf8fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x64bf8fp8.json"},{"mnemonic":"v_smfmac_f32_32x32x64fp8bf8","slug":"v_smfmac_f32_32x32x64fp8bf8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x64fp8bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x64fp8bf8.json"},{"mnemonic":"v_smfmac_f32_32x32x64fp8fp8","slug":"v_smfmac_f32_32x32x64fp8fp8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_f32_32x32x64fp8fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_f32_32x32x64fp8fp8.json"},{"mnemonic":"v_smfmac_i32_16x16x128_i8","slug":"v_smfmac_i32_16x16x128_i8","records":1,"summary":"Multiply the 16x128 sparse matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix stored…","page":"https://instructionsets.com/amdgpu/v_smfmac_i32_16x16x128_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_i32_16x16x128_i8.json"},{"mnemonic":"v_smfmac_i32_16x16x128i8","slug":"v_smfmac_i32_16x16x128i8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_i32_16x16x128i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_i32_16x16x128i8.json"},{"mnemonic":"v_smfmac_i32_16x16x64_i8","slug":"v_smfmac_i32_16x16x64_i8","records":1,"summary":"Multiply the 16x64 sparse matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_i32_16x16x64_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_i32_16x16x64_i8.json","aliases":["v_smfmac_i32_16x16x64i8"]},{"mnemonic":"v_smfmac_i32_16x16x64i8","slug":"v_smfmac_i32_16x16x64i8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_i32_16x16x64i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_i32_16x16x64i8.json","aliases":["v_smfmac_i32_16x16x64_i8"]},{"mnemonic":"v_smfmac_i32_32x32x32_i8","slug":"v_smfmac_i32_32x32x32_i8","records":1,"summary":"Multiply the 32x32 sparse matrix in the first input by the 32x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_i32_32x32x32_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_i32_32x32x32_i8.json","aliases":["v_smfmac_i32_32x32x32i8"]},{"mnemonic":"v_smfmac_i32_32x32x32i8","slug":"v_smfmac_i32_32x32x32i8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_i32_32x32x32i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_i32_32x32x32i8.json","aliases":["v_smfmac_i32_32x32x32_i8"]},{"mnemonic":"v_smfmac_i32_32x32x64_i8","slug":"v_smfmac_i32_32x32x64_i8","records":1,"summary":"Multiply the 32x64 sparse matrix in the first input by the 64x32 matrix in the second input and accumulate the result into the 32x32 matrix stored in…","page":"https://instructionsets.com/amdgpu/v_smfmac_i32_32x32x64_i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_i32_32x32x64_i8.json"},{"mnemonic":"v_smfmac_i32_32x32x64i8","slug":"v_smfmac_i32_32x32x64i8","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_smfmac_i32_32x32x64i8/","api":"https://instructionsets.com/api/v1/amdgpu/v_smfmac_i32_32x32x64i8.json"},{"mnemonic":"v_sqrt_bf16","slug":"v_sqrt_bf16","records":1,"summary":"AMDGPU VOP1 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_sqrt_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sqrt_bf16.json"},{"mnemonic":"v_sqrt_f16","slug":"v_sqrt_f16","records":1,"summary":"Calculate the square root of the half-precision float input using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_sqrt_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sqrt_f16.json"},{"mnemonic":"v_sqrt_f32","slug":"v_sqrt_f32","records":1,"summary":"Calculate the square root of the single-precision float input using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_sqrt_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sqrt_f32.json"},{"mnemonic":"v_sqrt_f64","slug":"v_sqrt_f64","records":1,"summary":"Calculate the square root of the double-precision float input using IEEE rules and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_sqrt_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_sqrt_f64.json"},{"mnemonic":"v_sub_co_ci_u32","slug":"v_sub_co_ci_u32","records":1,"summary":"Subtract the second unsigned 32-bit integer input from the first input, subtract a bit from the carry-in mask, store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_sub_co_ci_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_co_ci_u32.json"},{"mnemonic":"v_sub_co_u32","slug":"v_sub_co_u32","records":1,"summary":"Subtract the second unsigned 32-bit integer input from the first input, store the result into a vector register and store the carry-out mask into a…","page":"https://instructionsets.com/amdgpu/v_sub_co_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_co_u32.json"},{"mnemonic":"v_sub_f16","slug":"v_sub_f16","records":1,"summary":"Subtract the second floating point input from the first input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_sub_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_f16.json"},{"mnemonic":"v_sub_f32","slug":"v_sub_f32","records":1,"summary":"Per-lane single-precision floating-point subtract.","page":"https://instructionsets.com/amdgpu/v_sub_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_f32.json"},{"mnemonic":"v_sub_i16","slug":"v_sub_i16","records":1,"summary":"Subtract the second signed 16-bit integer input from the first input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_sub_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_i16.json"},{"mnemonic":"v_sub_i32","slug":"v_sub_i32","records":1,"summary":"Subtract the second signed 32-bit integer input from the first input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_sub_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_i32.json"},{"mnemonic":"v_sub_nc_i16","slug":"v_sub_nc_i16","records":1,"summary":"Subtract the second signed 16-bit integer input from the first input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_sub_nc_i16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_nc_i16.json"},{"mnemonic":"v_sub_nc_i32","slug":"v_sub_nc_i32","records":1,"summary":"Subtract the second signed 32-bit integer input from the first input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_sub_nc_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_nc_i32.json"},{"mnemonic":"v_sub_nc_u16","slug":"v_sub_nc_u16","records":1,"summary":"Subtract the second unsigned 16-bit integer input from the first input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_sub_nc_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_nc_u16.json"},{"mnemonic":"v_sub_nc_u32","slug":"v_sub_nc_u32","records":1,"summary":"Subtract the second unsigned 32-bit integer input from the first input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_sub_nc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_nc_u32.json"},{"mnemonic":"v_sub_nc_u64","slug":"v_sub_nc_u64","records":1,"summary":"AMDGPU VOP2 vector instruction operating on u64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_sub_nc_u64/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_nc_u64.json"},{"mnemonic":"v_sub_u16","slug":"v_sub_u16","records":1,"summary":"Subtract the second unsigned 16-bit integer input from the first input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_sub_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_u16.json"},{"mnemonic":"v_sub_u32","slug":"v_sub_u32","records":1,"summary":"Subtract the second unsigned 32-bit integer input from the first input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_sub_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_sub_u32.json"},{"mnemonic":"v_subb_co_u32","slug":"v_subb_co_u32","records":1,"summary":"Subtract the second unsigned 32-bit integer input from the first input, subtract a bit from the carry-in mask, store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_subb_co_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subb_co_u32.json"},{"mnemonic":"v_subb_u32","slug":"v_subb_u32","records":1,"summary":"AMDGPU VOP2 vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_subb_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subb_u32.json"},{"mnemonic":"v_subbrev_co_u32","slug":"v_subbrev_co_u32","records":1,"summary":"Subtract the first unsigned 32-bit integer input from the second input, subtract a bit from the carry-in mask, store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_subbrev_co_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subbrev_co_u32.json"},{"mnemonic":"v_subbrev_u32","slug":"v_subbrev_u32","records":1,"summary":"AMDGPU VOP2 vector instruction operating on u32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_subbrev_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subbrev_u32.json"},{"mnemonic":"v_subrev_co_ci_u32","slug":"v_subrev_co_ci_u32","records":1,"summary":"Subtract the first unsigned 32-bit integer input from the second input, subtract a bit from the carry-in mask, store the result into a vector…","page":"https://instructionsets.com/amdgpu/v_subrev_co_ci_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subrev_co_ci_u32.json"},{"mnemonic":"v_subrev_co_u32","slug":"v_subrev_co_u32","records":1,"summary":"Subtract the first unsigned 32-bit integer input from the second input, store the result into a vector register and store the carry-out mask into a…","page":"https://instructionsets.com/amdgpu/v_subrev_co_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subrev_co_u32.json"},{"mnemonic":"v_subrev_f16","slug":"v_subrev_f16","records":1,"summary":"Subtract the first floating point input from the second input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_subrev_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_subrev_f16.json"},{"mnemonic":"v_subrev_f32","slug":"v_subrev_f32","records":1,"summary":"Subtract the first floating point input from the second input and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_subrev_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subrev_f32.json"},{"mnemonic":"v_subrev_i32","slug":"v_subrev_i32","records":1,"summary":"AMDGPU VOP2 vector instruction operating on i32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_subrev_i32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subrev_i32.json"},{"mnemonic":"v_subrev_nc_u32","slug":"v_subrev_nc_u32","records":1,"summary":"Subtract the first unsigned 32-bit integer input from the second input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_subrev_nc_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subrev_nc_u32.json"},{"mnemonic":"v_subrev_u16","slug":"v_subrev_u16","records":1,"summary":"Subtract the first unsigned 16-bit integer input from the second input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_subrev_u16/","api":"https://instructionsets.com/api/v1/amdgpu/v_subrev_u16.json"},{"mnemonic":"v_subrev_u32","slug":"v_subrev_u32","records":1,"summary":"Subtract the first unsigned 32-bit integer input from the second input and store the result into a vector register. No carry-in or carry-out support.","page":"https://instructionsets.com/amdgpu/v_subrev_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_subrev_u32.json"},{"mnemonic":"v_swap_b16","slug":"v_swap_b16","records":1,"summary":"Swap the values in two vector registers.","page":"https://instructionsets.com/amdgpu/v_swap_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swap_b16.json"},{"mnemonic":"v_swap_b32","slug":"v_swap_b32","records":1,"summary":"Swap the values in two vector registers.","page":"https://instructionsets.com/amdgpu/v_swap_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_swap_b32.json"},{"mnemonic":"v_swaprel_b32","slug":"v_swaprel_b32","records":1,"summary":"Swap the values in two relatively-indexed vector registers.","page":"https://instructionsets.com/amdgpu/v_swaprel_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_swaprel_b32.json"},{"mnemonic":"v_swmmac_bf16_16x16x32_bf16","slug":"v_swmmac_bf16_16x16x32_bf16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_bf16_16x16x32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_bf16_16x16x32_bf16.json"},{"mnemonic":"v_swmmac_bf16_16x16x64_bf16","slug":"v_swmmac_bf16_16x16x64_bf16","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_bf16_16x16x64_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_bf16_16x16x64_bf16.json"},{"mnemonic":"v_swmmac_bf16f32_16x16x64_bf16","slug":"v_swmmac_bf16f32_16x16x64_bf16","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_bf16f32_16x16x64_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_bf16f32_16x16x64_bf16.json"},{"mnemonic":"v_swmmac_f16_16x16x128_bf8_bf8","slug":"v_swmmac_f16_16x16x128_bf8_bf8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f16_16x16x128_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f16_16x16x128_bf8_bf8.json"},{"mnemonic":"v_swmmac_f16_16x16x128_bf8_fp8","slug":"v_swmmac_f16_16x16x128_bf8_fp8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f16_16x16x128_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f16_16x16x128_bf8_fp8.json"},{"mnemonic":"v_swmmac_f16_16x16x128_fp8_bf8","slug":"v_swmmac_f16_16x16x128_fp8_bf8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f16_16x16x128_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f16_16x16x128_fp8_bf8.json"},{"mnemonic":"v_swmmac_f16_16x16x128_fp8_fp8","slug":"v_swmmac_f16_16x16x128_fp8_fp8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f16_16x16x128_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f16_16x16x128_fp8_fp8.json"},{"mnemonic":"v_swmmac_f16_16x16x32_f16","slug":"v_swmmac_f16_16x16x32_f16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f16_16x16x32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f16_16x16x32_f16.json"},{"mnemonic":"v_swmmac_f16_16x16x64_f16","slug":"v_swmmac_f16_16x16x64_f16","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f16_16x16x64_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f16_16x16x64_f16.json"},{"mnemonic":"v_swmmac_f32_16x16x128_bf8_bf8","slug":"v_swmmac_f32_16x16x128_bf8_bf8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x128_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x128_bf8_bf8.json"},{"mnemonic":"v_swmmac_f32_16x16x128_bf8_fp8","slug":"v_swmmac_f32_16x16x128_bf8_fp8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x128_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x128_bf8_fp8.json"},{"mnemonic":"v_swmmac_f32_16x16x128_fp8_bf8","slug":"v_swmmac_f32_16x16x128_fp8_bf8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x128_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x128_fp8_bf8.json"},{"mnemonic":"v_swmmac_f32_16x16x128_fp8_fp8","slug":"v_swmmac_f32_16x16x128_fp8_fp8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x128_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x128_fp8_fp8.json"},{"mnemonic":"v_swmmac_f32_16x16x32_bf16","slug":"v_swmmac_f32_16x16x32_bf16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x32_bf16.json"},{"mnemonic":"v_swmmac_f32_16x16x32_bf8_bf8","slug":"v_swmmac_f32_16x16x32_bf8_bf8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x32_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x32_bf8_bf8.json"},{"mnemonic":"v_swmmac_f32_16x16x32_bf8_fp8","slug":"v_swmmac_f32_16x16x32_bf8_fp8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x32_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x32_bf8_fp8.json"},{"mnemonic":"v_swmmac_f32_16x16x32_f16","slug":"v_swmmac_f32_16x16x32_f16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x32_f16.json"},{"mnemonic":"v_swmmac_f32_16x16x32_fp8_bf8","slug":"v_swmmac_f32_16x16x32_fp8_bf8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x32_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x32_fp8_bf8.json"},{"mnemonic":"v_swmmac_f32_16x16x32_fp8_fp8","slug":"v_swmmac_f32_16x16x32_fp8_fp8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x32_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x32_fp8_fp8.json"},{"mnemonic":"v_swmmac_f32_16x16x64_bf16","slug":"v_swmmac_f32_16x16x64_bf16","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x64_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x64_bf16.json"},{"mnemonic":"v_swmmac_f32_16x16x64_f16","slug":"v_swmmac_f32_16x16x64_f16","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_f32_16x16x64_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_f32_16x16x64_f16.json"},{"mnemonic":"v_swmmac_i32_16x16x128_iu8","slug":"v_swmmac_i32_16x16x128_iu8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_i32_16x16x128_iu8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_i32_16x16x128_iu8.json"},{"mnemonic":"v_swmmac_i32_16x16x32_iu4","slug":"v_swmmac_i32_16x16x32_iu4","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_i32_16x16x32_iu4/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_i32_16x16x32_iu4.json"},{"mnemonic":"v_swmmac_i32_16x16x32_iu8","slug":"v_swmmac_i32_16x16x32_iu8","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_i32_16x16x32_iu8/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_i32_16x16x32_iu8.json"},{"mnemonic":"v_swmmac_i32_16x16x64_iu4","slug":"v_swmmac_i32_16x16x64_iu4","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and accumulate the result into the 16x16 matrix in the…","page":"https://instructionsets.com/amdgpu/v_swmmac_i32_16x16x64_iu4/","api":"https://instructionsets.com/api/v1/amdgpu/v_swmmac_i32_16x16x64_iu4.json"},{"mnemonic":"v_tanh_bf16","slug":"v_tanh_bf16","records":1,"summary":"AMDGPU VOP1 vector instruction. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_tanh_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_tanh_bf16.json"},{"mnemonic":"v_tanh_f16","slug":"v_tanh_f16","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_tanh_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_tanh_f16.json"},{"mnemonic":"v_tanh_f32","slug":"v_tanh_f32","records":1,"summary":"AMDGPU VOP1 vector instruction operating on f32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_tanh_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_tanh_f32.json"},{"mnemonic":"v_trig_preop_f64","slug":"v_trig_preop_f64","records":1,"summary":"Look up a 53-bit segment of 2/PI using an integer segment select in the second input.","page":"https://instructionsets.com/amdgpu/v_trig_preop_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_trig_preop_f64.json"},{"mnemonic":"v_trunc_f16","slug":"v_trunc_f16","records":1,"summary":"Compute the integer part of a half-precision float input using round toward zero semantics and store the result in floating point format into a…","page":"https://instructionsets.com/amdgpu/v_trunc_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_trunc_f16.json"},{"mnemonic":"v_trunc_f32","slug":"v_trunc_f32","records":1,"summary":"Compute the integer part of a single-precision float input using round toward zero semantics and store the result in floating point format into a…","page":"https://instructionsets.com/amdgpu/v_trunc_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_trunc_f32.json"},{"mnemonic":"v_trunc_f64","slug":"v_trunc_f64","records":1,"summary":"Compute the integer part of a double-precision float input using round toward zero semantics and store the result in floating point format into a…","page":"https://instructionsets.com/amdgpu/v_trunc_f64/","api":"https://instructionsets.com/api/v1/amdgpu/v_trunc_f64.json"},{"mnemonic":"v_wmma_bf16_16x16x16_bf16","slug":"v_wmma_bf16_16x16x16_bf16","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_bf16_16x16x16_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_bf16_16x16x16_bf16.json"},{"mnemonic":"v_wmma_bf16_16x16x32_bf16","slug":"v_wmma_bf16_16x16x32_bf16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_bf16_16x16x32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_bf16_16x16x32_bf16.json"},{"mnemonic":"v_wmma_bf16f32_16x16x32_bf16","slug":"v_wmma_bf16f32_16x16x32_bf16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_bf16f32_16x16x32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_bf16f32_16x16x32_bf16.json"},{"mnemonic":"v_wmma_f16_16x16x128_bf8_bf8","slug":"v_wmma_f16_16x16x128_bf8_bf8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x128_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x128_bf8_bf8.json"},{"mnemonic":"v_wmma_f16_16x16x128_bf8_fp8","slug":"v_wmma_f16_16x16x128_bf8_fp8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x128_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x128_bf8_fp8.json"},{"mnemonic":"v_wmma_f16_16x16x128_fp8_bf8","slug":"v_wmma_f16_16x16x128_fp8_bf8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x128_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x128_fp8_bf8.json"},{"mnemonic":"v_wmma_f16_16x16x128_fp8_fp8","slug":"v_wmma_f16_16x16x128_fp8_fp8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x128_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x128_fp8_fp8.json"},{"mnemonic":"v_wmma_f16_16x16x16_f16","slug":"v_wmma_f16_16x16x16_f16","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x16_f16.json"},{"mnemonic":"v_wmma_f16_16x16x32_f16","slug":"v_wmma_f16_16x16x32_f16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x32_f16.json"},{"mnemonic":"v_wmma_f16_16x16x64_bf8_bf8","slug":"v_wmma_f16_16x16x64_bf8_bf8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x64_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x64_bf8_bf8.json"},{"mnemonic":"v_wmma_f16_16x16x64_bf8_fp8","slug":"v_wmma_f16_16x16x64_bf8_fp8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x64_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x64_bf8_fp8.json"},{"mnemonic":"v_wmma_f16_16x16x64_fp8_bf8","slug":"v_wmma_f16_16x16x64_fp8_bf8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x64_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x64_fp8_bf8.json"},{"mnemonic":"v_wmma_f16_16x16x64_fp8_fp8","slug":"v_wmma_f16_16x16x64_fp8_fp8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f16_16x16x64_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f16_16x16x64_fp8_fp8.json"},{"mnemonic":"v_wmma_f32_16x16x128_bf8_bf8","slug":"v_wmma_f32_16x16x128_bf8_bf8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x128_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x128_bf8_bf8.json"},{"mnemonic":"v_wmma_f32_16x16x128_bf8_fp8","slug":"v_wmma_f32_16x16x128_bf8_fp8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x128_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x128_bf8_fp8.json"},{"mnemonic":"v_wmma_f32_16x16x128_f8f6f4","slug":"v_wmma_f32_16x16x128_f8f6f4","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x128_f8f6f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x128_f8f6f4.json"},{"mnemonic":"v_wmma_f32_16x16x128_fp8_bf8","slug":"v_wmma_f32_16x16x128_fp8_bf8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x128_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x128_fp8_bf8.json"},{"mnemonic":"v_wmma_f32_16x16x128_fp8_fp8","slug":"v_wmma_f32_16x16x128_fp8_fp8","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x128_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x128_fp8_fp8.json"},{"mnemonic":"v_wmma_f32_16x16x16_bf16","slug":"v_wmma_f32_16x16x16_bf16","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x16_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x16_bf16.json"},{"mnemonic":"v_wmma_f32_16x16x16_bf8_bf8","slug":"v_wmma_f32_16x16x16_bf8_bf8","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x16_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x16_bf8_bf8.json"},{"mnemonic":"v_wmma_f32_16x16x16_bf8_fp8","slug":"v_wmma_f32_16x16x16_bf8_fp8","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x16_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x16_bf8_fp8.json"},{"mnemonic":"v_wmma_f32_16x16x16_f16","slug":"v_wmma_f32_16x16x16_f16","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x16_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x16_f16.json"},{"mnemonic":"v_wmma_f32_16x16x16_fp8_bf8","slug":"v_wmma_f32_16x16x16_fp8_bf8","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x16_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x16_fp8_bf8.json"},{"mnemonic":"v_wmma_f32_16x16x16_fp8_fp8","slug":"v_wmma_f32_16x16x16_fp8_fp8","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x16_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x16_fp8_fp8.json"},{"mnemonic":"v_wmma_f32_16x16x32_bf16","slug":"v_wmma_f32_16x16x32_bf16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x32_bf16/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x32_bf16.json"},{"mnemonic":"v_wmma_f32_16x16x32_f16","slug":"v_wmma_f32_16x16x32_f16","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x32_f16/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x32_f16.json"},{"mnemonic":"v_wmma_f32_16x16x4_f32","slug":"v_wmma_f32_16x16x4_f32","records":1,"summary":"Multiply the 16x4 matrix in the first input by the 4x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x4_f32/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x4_f32.json"},{"mnemonic":"v_wmma_f32_16x16x64_bf8_bf8","slug":"v_wmma_f32_16x16x64_bf8_bf8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x64_bf8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x64_bf8_bf8.json"},{"mnemonic":"v_wmma_f32_16x16x64_bf8_fp8","slug":"v_wmma_f32_16x16x64_bf8_fp8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x64_bf8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x64_bf8_fp8.json"},{"mnemonic":"v_wmma_f32_16x16x64_fp8_bf8","slug":"v_wmma_f32_16x16x64_fp8_bf8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x64_fp8_bf8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x64_fp8_bf8.json"},{"mnemonic":"v_wmma_f32_16x16x64_fp8_fp8","slug":"v_wmma_f32_16x16x64_fp8_fp8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_16x16x64_fp8_fp8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_16x16x64_fp8_fp8.json"},{"mnemonic":"v_wmma_f32_32x16x128_f4","slug":"v_wmma_f32_32x16x128_f4","records":1,"summary":"Multiply the 32x128 matrix in the first input by the 128x16 matrix in the second input and add the 32x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_f32_32x16x128_f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_f32_32x16x128_f4.json"},{"mnemonic":"v_wmma_i32_16x16x16_iu4","slug":"v_wmma_i32_16x16x16_iu4","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_i32_16x16x16_iu4/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_i32_16x16x16_iu4.json"},{"mnemonic":"v_wmma_i32_16x16x16_iu8","slug":"v_wmma_i32_16x16x16_iu8","records":1,"summary":"Multiply the 16x16 matrix in the first input by the 16x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_i32_16x16x16_iu8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_i32_16x16x16_iu8.json"},{"mnemonic":"v_wmma_i32_16x16x32_iu4","slug":"v_wmma_i32_16x16x32_iu4","records":1,"summary":"Multiply the 16x32 matrix in the first input by the 32x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_i32_16x16x32_iu4/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_i32_16x16x32_iu4.json"},{"mnemonic":"v_wmma_i32_16x16x64_iu8","slug":"v_wmma_i32_16x16x64_iu8","records":1,"summary":"Multiply the 16x64 matrix in the first input by the 64x16 matrix in the second input and add the 16x16 matrix in the third input using fused multiply…","page":"https://instructionsets.com/amdgpu/v_wmma_i32_16x16x64_iu8/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_i32_16x16x64_iu8.json"},{"mnemonic":"v_wmma_ld_scale16_paired_b64","slug":"v_wmma_ld_scale16_paired_b64","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on b64 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_wmma_ld_scale16_paired_b64/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_ld_scale16_paired_b64.json"},{"mnemonic":"v_wmma_ld_scale_paired_b32","slug":"v_wmma_ld_scale_paired_b32","records":1,"summary":"AMDGPU VOP3P matrix instruction operating on b32 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_wmma_ld_scale_paired_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_ld_scale_paired_b32.json"},{"mnemonic":"v_wmma_scale16_f32_16x16x128_f8f6f4","slug":"v_wmma_scale16_f32_16x16x128_f8f6f4","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_scale16_f32_16x16x128_f8f6f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_scale16_f32_16x16x128_f8f6f4.json"},{"mnemonic":"v_wmma_scale16_f32_32x16x128_f4","slug":"v_wmma_scale16_f32_32x16x128_f4","records":1,"summary":"Multiply the 32x128 matrix in the first input by the 128x16 matrix in the second input and add the 32x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_scale16_f32_32x16x128_f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_scale16_f32_32x16x128_f4.json"},{"mnemonic":"v_wmma_scale_f32_16x16x128_f8f6f4","slug":"v_wmma_scale_f32_16x16x128_f8f6f4","records":1,"summary":"Multiply the 16x128 matrix in the first input by the 128x16 matrix in the second input and add the 16x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_scale_f32_16x16x128_f8f6f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_scale_f32_16x16x128_f8f6f4.json"},{"mnemonic":"v_wmma_scale_f32_32x16x128_f4","slug":"v_wmma_scale_f32_32x16x128_f4","records":1,"summary":"Multiply the 32x128 matrix in the first input by the 128x16 matrix in the second input and add the 32x16 matrix in the third input using fused…","page":"https://instructionsets.com/amdgpu/v_wmma_scale_f32_32x16x128_f4/","api":"https://instructionsets.com/api/v1/amdgpu/v_wmma_scale_f32_32x16x128_f4.json"},{"mnemonic":"v_writelane_b32","slug":"v_writelane_b32","records":1,"summary":"Write the scalar value in the first input into the specified lane of a vector register where the lane select is in the second input.","page":"https://instructionsets.com/amdgpu/v_writelane_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_writelane_b32.json"},{"mnemonic":"v_xad_u32","slug":"v_xad_u32","records":1,"summary":"Calculate bitwise XOR of the first two vector inputs, then add the third vector input to the intermediate result, then store the final result into a…","page":"https://instructionsets.com/amdgpu/v_xad_u32/","api":"https://instructionsets.com/api/v1/amdgpu/v_xad_u32.json","aliases":["v_xor_add_u32"]},{"mnemonic":"v_xnor_b32","slug":"v_xnor_b32","records":1,"summary":"Calculate bitwise XNOR on two vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_xnor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_xnor_b32.json"},{"mnemonic":"v_xor3_b32","slug":"v_xor3_b32","records":1,"summary":"Calculate the bitwise XOR of three vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_xor3_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_xor3_b32.json"},{"mnemonic":"v_xor_b16","slug":"v_xor_b16","records":1,"summary":"Calculate bitwise XOR on two vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_xor_b16/","api":"https://instructionsets.com/api/v1/amdgpu/v_xor_b16.json"},{"mnemonic":"v_xor_b16_fake16","slug":"v_xor_b16_fake16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_xor_b16_fake16/","api":"https://instructionsets.com/api/v1/amdgpu/v_xor_b16_fake16.json"},{"mnemonic":"v_xor_b16_t16","slug":"v_xor_b16_t16","records":1,"summary":"AMDGPU VOP2 vector instruction operating on b16 data. (Format and name extracted from LLVM's AMDGPU backend source - semantics not yet curated.)","page":"https://instructionsets.com/amdgpu/v_xor_b16_t16/","api":"https://instructionsets.com/api/v1/amdgpu/v_xor_b16_t16.json"},{"mnemonic":"v_xor_b32","slug":"v_xor_b32","records":1,"summary":"Calculate bitwise XOR on two vector inputs and store the result into a vector register.","page":"https://instructionsets.com/amdgpu/v_xor_b32/","api":"https://instructionsets.com/api/v1/amdgpu/v_xor_b32.json"}]}
