//clika-runtime/io.clika.runtime/Ops/scaledDotProductAttention
scaledDotProductAttention
[common]
fun scaledDotProductAttention(query: Tensor, key: Tensor, value: Tensor, attnMask: Tensor? = null, isCausal: Boolean = false, qScale: Tensor? = null, kScale: Tensor? = null, vScale: Tensor? = null): Tensor
scaledDotProductAttention(query: Tensor, key: Tensor, value: Tensor, attnMask: Tensor? = null, isCausal: Boolean = false, qScale: Tensor? = null, kScale: Tensor? = null, vScale: Tensor? = null): the scaled_dot_product_attention operator. Scaled dot-product attention over dense head-major tensors. s defaults to (the q/k head size) and is replaced wholesale by q_scale when given. Layout is head-major, D innermost.