Skip to main content

//clika-runtime/io.clika.runtime/Ops/attentionVarlen

attentionVarlen

[common]
fun attentionVarlen(query: Tensor, key: Tensor, value: Tensor, cuSeqlensQ: Tensor, cuSeqlensK: Tensor, maxSeqlenQ: Tensor? = null, maxSeqlenK: Tensor? = null, attnMask: Tensor? = null, headSink: Tensor? = null, isCausal: Boolean? = null, qScale: Tensor? = null, softcap: Double? = null, slidingWindow: Long? = null, smoothSoftmax: Boolean? = null, kScale: Tensor? = null, vScale: Tensor? = null, keptPrefix: Tensor? = null, kvPositionOffset: Tensor? = null): Tensor

attentionVarlen(query: Tensor, key: Tensor, value: Tensor, cuSeqlensQ: Tensor, cuSeqlensK: Tensor, maxSeqlenQ: Tensor? = null, maxSeqlenK: Tensor? = null, attnMask: Tensor? = null, headSink: Tensor? = null, isCausal: Boolean? = null, qScale: Tensor? = null, softcap: Double? = null, slidingWindow: Long? = null, smoothSoftmax: Boolean? = null, kScale: Tensor? = null, vScale: Tensor? = null, keptPrefix: Tensor? = null, kvPositionOffset: Tensor? = null): the attention_varlen operator. Variable-length (packed) form of attention: the serving riders over token-packed ragged batches. Tensors and offsets follow scaled_dot_product_attention_varlen; the riders (head_sink, softcap, sliding_window, smooth_softmax) follow attention.