//clika-runtime/io.clika.runtime/Ops/attention
attention
[common]
fun attention(query: Tensor, key: Tensor, value: Tensor, attnMask: Tensor? = null, headSink: Tensor? = null, isCausal: Boolean? = null, qScale: Tensor? = null, softcap: Double? = null, slidingWindow: Long? = null, smoothSoftmax: Boolean? = null, kScale: Tensor? = null, vScale: Tensor? = null, keptPrefix: Tensor? = null): Tensor
attention(query: Tensor, key: Tensor, value: Tensor, attnMask: Tensor? = null, headSink: Tensor? = null, isCausal: Boolean? = null, qScale: Tensor? = null, softcap: Double? = null, slidingWindow: Long? = null, smoothSoftmax: Boolean? = null, kScale: Tensor? = null, vScale: Tensor? = null, keptPrefix: Tensor? = null): the attention operator. Dense attention with the serving riders: per-head sink, logit soft-cap, sliding window, smoothed softmax. The core is scaled_dot_product_attention; each rider adjusts the softmax stage: - head_sink``[H_q]: a per-head virtual logit folded into the softmax denominator (attention that can "go nowhere"). - softcap: logits pass through cap * tanh(x / cap) before the softmax. - sliding_window: each query attends only the last N key positions. - smooth_softmax: adds one to the softmax denominator. Default false.