424def streamingAttentionForwardGroupedUncheckedSM86 = (lambda unrestricted seq : Nat . (lambda unrestricted heads : Nat . (lambda unrestricted keyValueHeads : Nat .
425 (let unrestricted groups = (naturalDivideUnchecked heads keyValueHeads) in
426 (let unrestricted headBytes = (naturalMultiply seq (naturalMultiply saHeadWidth saHalfBytes)) in
427 (let unrestricted rowBytes = (naturalMultiply heads (naturalMultiply saHeadWidth saHalfBytes)) in
428 (let unrestricted body = (saTileBody seq sm86ProgramEmpty) in
429 (saS2R saTid (constructor SM86SpecialRegister SM86ThreadIdX)
430 (saS2R saTileIndex (constructor SM86SpecialRegister SM86CooperativeThreadArrayIdX)
431 (saS2R saKeyValueHead (constructor SM86SpecialRegister SM86CooperativeThreadArrayIdY)
432 (saS2R saHead (constructor SM86SpecialRegister SM86CooperativeThreadArrayIdZ)
433 (saMovImm saOne 1
434 -- the special registers are in once this waits
435 (saMovImmAfter saScratch 31 saWait5
436 (saImad saHead saKeyValueHead groups saHead
437 (saAnd saLane saTid saScratch
438 (saShr saWarp saTid 5
439 (saShr saQuad saLane 2
440 (saMovImm saScratch 3
441 (saAnd saQuadLane saLane saScratch
442 -- the thread's query row: 64 tile + 16 warp + lane / 4
443 (saImad saRow saWarp 16 saQuad
444 (saImad saRow saTileIndex saTile saRow
445 -- Q: head h's plane, the row, the quad lane's pair of columns
446 (saMovImm saScratch 0
447 (saImad saScratch saRow (naturalMultiply saHeadWidth saHalfBytes) saScratch
448 (saImad saScratch saQuadLane 4 saScratch
449 (saImad saScratch saHead headBytes saScratch
450 (saWide saPointer saScratch saOne (saArgument 0)
451 (saFor 16 (lambda unrestricted i : Nat .
452 (let unrestricted kt = (naturalDivideUnchecked i 4) in (let unrestricted j = (naturalModuloUnchecked i 4) in
453 (saLoad (saQReg kt j) saPointer
454 (naturalAdd (naturalMultiply kt 32)
455 (naturalAdd (naturalMultiply (naturalModuloUnchecked j 2) (naturalMultiply 8 (naturalMultiply saHeadWidth saHalfBytes)))
456 (naturalMultiply (naturalDivideUnchecked j 2) 16)))
457 saSB1 saWaitNone))))
458 -- the output's offset in the merged plane (kept in saScratch): the row,
459 -- head h's 64 columns, the quad lane's pair
460 (saMovImm saScratch 0
461 (saImad saScratch saRow rowBytes saScratch
462 (saImad saScratch saHead (naturalMultiply saHeadWidth saHalfBytes) saScratch
463 (saImad saScratch saQuadLane 4 saScratch
464 -- the log-sum-exp's (kept in saRow): 4 (h seq + row)
465 (saImad saRow saHead seq saRow
466 (saMovImm saKeyOffset 0
467 (saImad saRow saRow 4 saKeyOffset
468 -- K from key 0: the quad's row lane / 4, its pair of columns
469 (saImad saKeyOffset saQuad (naturalMultiply saHeadWidth saHalfBytes) saKeyOffset
470 (saImad saKeyOffset saQuadLane 4 saKeyOffset
471 (saImad saKeyOffset saKeyValueHead headBytes saKeyOffset
472 -- V^T from key 0: row lane / 4 of head h's 64 x seq plane
473 (saMovImm saValueOffset 0
474 (saImad saValueOffset saQuad (naturalMultiply seq saHalfBytes) saValueOffset
475 (saImad saValueOffset saQuadLane 4 saValueOffset
476 (saImad saValueOffset saKeyValueHead headBytes saValueOffset
477 -- the tiles up to the diagonal one; the mask's base, 16 warp + lane / 4
478 -- - 2 (lane % 4) + 64 - 4096
479 (saAddImm saCount saTileIndex 1
480 (saImad saMaskBase saWarp 16 saQuad
481 (saImad saMaskBase saQuadLane 4294967294 saMaskBase
482 (saAddImm saMaskBase saMaskBase (naturalSaturatingSubtract 4294967296 (naturalSaturatingSubtract 4096 64))
483 (saMovConst saScale (saArgument 5)
484 (saMovImm (saM 0) saMinusInfinity (saMovImm (saM 1) saMinusInfinity
485 (saMovImm (saL 0) 0 (saMovImm (saL 1) 0
486 (saFor 32 (lambda unrestricted i : Nat . (saMovImm (naturalAdd 68 i) 0))
487 (sm86ProgramAppend body
488 (sm86ProgramAppend (saLoopTail seq (sm86ProgramCount body))
489 (saEpilogue seq heads
490 (saOp (constructor SM86InstructionBody SM86Exit (saAfter saWaitAll)) sm86ProgramEmpty)))))))))))))))))))))))))))))))))))))))))))))))))))))))The compiler supplied declaration spans and resolved links from this source snapshot. This page does not assert that this file belongs to a checked closure.