1#! /usr/bin/env perl 2# Copyright 2016-2026 The OpenSSL Project Authors. All Rights Reserved. 3# 4# Licensed under the Apache License 2.0 (the "License"). You may not use 5# this file except in compliance with the License. You can obtain a copy 6# in the file LICENSE in the source distribution or at 7# https://www.openssl.org/source/license.html 8 9# 10# ==================================================================== 11# Written by Andy Polyakov <appro@openssl.org> for the OpenSSL 12# project. The module is, however, dual licensed under OpenSSL and 13# CRYPTOGAMS licenses depending on where you obtain it. For further 14# details see http://www.openssl.org/~appro/cryptogams/. 15# ==================================================================== 16 17# March 2016 18# 19# Initial support for Fujitsu SPARC64 X/X+ comprises minimally 20# required key setup and single-block procedures. 21# 22# April 2016 23# 24# Add "teaser" CBC and CTR mode-specific subroutines. "Teaser" means 25# that parallelizable nature of CBC decrypt and CTR is not utilized 26# yet. CBC encrypt on the other hand is as good as it can possibly 27# get processing one byte in 4.1 cycles with 128-bit key on SPARC64 X. 28# This is ~6x faster than pure software implementation... 29# 30# July 2016 31# 32# Switch from faligndata to fshiftorx, which allows to omit alignaddr 33# instructions and improve single-block and short-input performance 34# with misaligned data. 35 36$output = pop and open STDOUT,">$output"; 37 38{ 39my ($inp,$out,$key,$rounds,$tmp,$mask) = map("%o$_",(0..5)); 40 41$code.=<<___; 42#ifndef __ASSEMBLER__ 43# define __ASSEMBLER__ 1 44#endif 45#include "crypto/sparc_arch.h" 46 47#define LOCALS (STACK_BIAS+STACK_FRAME) 48 49.text 50 51.globl aes_fx_encrypt 52.align 32 53aes_fx_encrypt: 54 and $inp, 7, $tmp ! is input aligned? 55 andn $inp, 7, $inp 56 ldd [$key + 0], %f6 ! round[0] 57 ldd [$key + 8], %f8 58 mov %o7, %g1 59 ld [$key + 240], $rounds 60 611: call .+8 62 add %o7, .Linp_align-1b, %o7 63 64 sll $tmp, 3, $tmp 65 ldd [$inp + 0], %f0 ! load input 66 brz,pt $tmp, .Lenc_inp_aligned 67 ldd [$inp + 8], %f2 68 69 ldd [%o7 + $tmp], %f14 ! shift left params 70 ldd [$inp + 16], %f4 71 fshiftorx %f0, %f2, %f14, %f0 72 fshiftorx %f2, %f4, %f14, %f2 73 74.Lenc_inp_aligned: 75 ldd [$key + 16], %f10 ! round[1] 76 ldd [$key + 24], %f12 77 78 fxor %f0, %f6, %f0 ! ^=round[0] 79 fxor %f2, %f8, %f2 80 ldd [$key + 32], %f6 ! round[2] 81 ldd [$key + 40], %f8 82 add $key, 32, $key 83 sub $rounds, 4, $rounds 84 85.Loop_enc: 86 fmovd %f0, %f4 87 faesencx %f2, %f10, %f0 88 faesencx %f4, %f12, %f2 89 ldd [$key + 16], %f10 90 ldd [$key + 24], %f12 91 add $key, 32, $key 92 93 fmovd %f0, %f4 94 faesencx %f2, %f6, %f0 95 faesencx %f4, %f8, %f2 96 ldd [$key + 0], %f6 97 ldd [$key + 8], %f8 98 99 brnz,a $rounds, .Loop_enc 100 sub $rounds, 2, $rounds 101 102 andcc $out, 7, $tmp ! is output aligned? 103 andn $out, 7, $out 104 mov 0xff, $mask 105 srl $mask, $tmp, $mask 106 add %o7, 64, %o7 107 sll $tmp, 3, $tmp 108 109 fmovd %f0, %f4 110 faesencx %f2, %f10, %f0 111 faesencx %f4, %f12, %f2 112 ldd [%o7 + $tmp], %f14 ! shift right params 113 114 fmovd %f0, %f4 115 faesenclx %f2, %f6, %f0 116 faesenclx %f4, %f8, %f2 117 118 bnz,pn %icc, .Lenc_out_unaligned 119 mov %g1, %o7 120 121 std %f0, [$out + 0] 122 retl 123 std %f2, [$out + 8] 124 125.align 16 126.Lenc_out_unaligned: 127 add $out, 16, $inp 128 orn %g0, $mask, $tmp 129 fshiftorx %f0, %f0, %f14, %f4 130 fshiftorx %f0, %f2, %f14, %f6 131 fshiftorx %f2, %f2, %f14, %f8 132 133 stda %f4, [$out + $mask]0xc0 ! partial store 134 std %f6, [$out + 8] 135 stda %f8, [$inp + $tmp]0xc0 ! partial store 136 retl 137 nop 138.type aes_fx_encrypt,#function 139.size aes_fx_encrypt,.-aes_fx_encrypt 140 141.globl aes_fx_decrypt 142.align 32 143aes_fx_decrypt: 144 and $inp, 7, $tmp ! is input aligned? 145 andn $inp, 7, $inp 146 ldd [$key + 0], %f6 ! round[0] 147 ldd [$key + 8], %f8 148 mov %o7, %g1 149 ld [$key + 240], $rounds 150 1511: call .+8 152 add %o7, .Linp_align-1b, %o7 153 154 sll $tmp, 3, $tmp 155 ldd [$inp + 0], %f0 ! load input 156 brz,pt $tmp, .Ldec_inp_aligned 157 ldd [$inp + 8], %f2 158 159 ldd [%o7 + $tmp], %f14 ! shift left params 160 ldd [$inp + 16], %f4 161 fshiftorx %f0, %f2, %f14, %f0 162 fshiftorx %f2, %f4, %f14, %f2 163 164.Ldec_inp_aligned: 165 ldd [$key + 16], %f10 ! round[1] 166 ldd [$key + 24], %f12 167 168 fxor %f0, %f6, %f0 ! ^=round[0] 169 fxor %f2, %f8, %f2 170 ldd [$key + 32], %f6 ! round[2] 171 ldd [$key + 40], %f8 172 add $key, 32, $key 173 sub $rounds, 4, $rounds 174 175.Loop_dec: 176 fmovd %f0, %f4 177 faesdecx %f2, %f10, %f0 178 faesdecx %f4, %f12, %f2 179 ldd [$key + 16], %f10 180 ldd [$key + 24], %f12 181 add $key, 32, $key 182 183 fmovd %f0, %f4 184 faesdecx %f2, %f6, %f0 185 faesdecx %f4, %f8, %f2 186 ldd [$key + 0], %f6 187 ldd [$key + 8], %f8 188 189 brnz,a $rounds, .Loop_dec 190 sub $rounds, 2, $rounds 191 192 andcc $out, 7, $tmp ! is output aligned? 193 andn $out, 7, $out 194 mov 0xff, $mask 195 srl $mask, $tmp, $mask 196 add %o7, 64, %o7 197 sll $tmp, 3, $tmp 198 199 fmovd %f0, %f4 200 faesdecx %f2, %f10, %f0 201 faesdecx %f4, %f12, %f2 202 ldd [%o7 + $tmp], %f14 ! shift right params 203 204 fmovd %f0, %f4 205 faesdeclx %f2, %f6, %f0 206 faesdeclx %f4, %f8, %f2 207 208 bnz,pn %icc, .Ldec_out_unaligned 209 mov %g1, %o7 210 211 std %f0, [$out + 0] 212 retl 213 std %f2, [$out + 8] 214 215.align 16 216.Ldec_out_unaligned: 217 add $out, 16, $inp 218 orn %g0, $mask, $tmp 219 fshiftorx %f0, %f0, %f14, %f4 220 fshiftorx %f0, %f2, %f14, %f6 221 fshiftorx %f2, %f2, %f14, %f8 222 223 stda %f4, [$out + $mask]0xc0 ! partial store 224 std %f6, [$out + 8] 225 stda %f8, [$inp + $tmp]0xc0 ! partial store 226 retl 227 nop 228.type aes_fx_decrypt,#function 229.size aes_fx_decrypt,.-aes_fx_decrypt 230___ 231} 232{ 233my ($inp,$bits,$out,$tmp,$inc) = map("%o$_",(0..5)); 234$code.=<<___; 235.globl aes_fx_set_decrypt_key 236.align 32 237aes_fx_set_decrypt_key: 238 b .Lset_encrypt_key 239 mov -1, $inc 240 retl 241 nop 242.type aes_fx_set_decrypt_key,#function 243.size aes_fx_set_decrypt_key,.-aes_fx_set_decrypt_key 244 245.globl aes_fx_set_encrypt_key 246.align 32 247aes_fx_set_encrypt_key: 248 mov 1, $inc 249 nop 250.Lset_encrypt_key: 251 and $inp, 7, $tmp 252 andn $inp, 7, $inp 253 sll $tmp, 3, $tmp 254 mov %o7, %g1 255 2561: call .+8 257 add %o7, .Linp_align-1b, %o7 258 259 ldd [%o7 + $tmp], %f10 ! shift left params 260 mov %g1, %o7 261 262 cmp $bits, 192 263 ldd [$inp + 0], %f0 264 bl,pt %icc, .L128 265 ldd [$inp + 8], %f2 266 267 be,pt %icc, .L192 268 ldd [$inp + 16], %f4 269 brz,pt $tmp, .L256aligned 270 ldd [$inp + 24], %f6 271 272 ldd [$inp + 32], %f8 273 fshiftorx %f0, %f2, %f10, %f0 274 fshiftorx %f2, %f4, %f10, %f2 275 fshiftorx %f4, %f6, %f10, %f4 276 fshiftorx %f6, %f8, %f10, %f6 277 278.L256aligned: 279 mov 14, $bits 280 and $inc, `14*16`, $tmp 281 st $bits, [$out + 240] ! store rounds 282 add $out, $tmp, $out ! start or end of key schedule 283 sllx $inc, 4, $inc ! 16 or -16 284___ 285for ($i=0; $i<6; $i++) { 286 $code.=<<___; 287 std %f0, [$out + 0] 288 faeskeyx %f6, `0x10+$i`, %f0 289 std %f2, [$out + 8] 290 add $out, $inc, $out 291 faeskeyx %f0, 0x00, %f2 292 std %f4, [$out + 0] 293 faeskeyx %f2, 0x01, %f4 294 std %f6, [$out + 8] 295 add $out, $inc, $out 296 faeskeyx %f4, 0x00, %f6 297___ 298} 299$code.=<<___; 300 std %f0, [$out + 0] 301 faeskeyx %f6, `0x10+$i`, %f0 302 std %f2, [$out + 8] 303 add $out, $inc, $out 304 faeskeyx %f0, 0x00, %f2 305 std %f4,[$out + 0] 306 std %f6,[$out + 8] 307 add $out, $inc, $out 308 std %f0,[$out + 0] 309 std %f2,[$out + 8] 310 retl 311 xor %o0, %o0, %o0 ! return 0 312 313.align 16 314.L192: 315 brz,pt $tmp, .L192aligned 316 nop 317 318 ldd [$inp + 24], %f6 319 fshiftorx %f0, %f2, %f10, %f0 320 fshiftorx %f2, %f4, %f10, %f2 321 fshiftorx %f4, %f6, %f10, %f4 322 323.L192aligned: 324 mov 12, $bits 325 and $inc, `12*16`, $tmp 326 st $bits, [$out + 240] ! store rounds 327 add $out, $tmp, $out ! start or end of key schedule 328 sllx $inc, 4, $inc ! 16 or -16 329___ 330for ($i=0; $i<8; $i+=2) { 331 $code.=<<___; 332 std %f0, [$out + 0] 333 faeskeyx %f4, `0x10+$i`, %f0 334 std %f2, [$out + 8] 335 add $out, $inc, $out 336 faeskeyx %f0, 0x00, %f2 337 std %f4, [$out + 0] 338 faeskeyx %f2, 0x00, %f4 339 std %f0, [$out + 8] 340 add $out, $inc, $out 341 faeskeyx %f4, `0x10+$i+1`, %f0 342 std %f2, [$out + 0] 343 faeskeyx %f0, 0x00, %f2 344 std %f4, [$out + 8] 345 add $out, $inc, $out 346___ 347$code.=<<___ if ($i<6); 348 faeskeyx %f2, 0x00, %f4 349___ 350} 351$code.=<<___; 352 std %f0, [$out + 0] 353 std %f2, [$out + 8] 354 retl 355 xor %o0, %o0, %o0 ! return 0 356 357.align 16 358.L128: 359 brz,pt $tmp, .L128aligned 360 nop 361 362 ldd [$inp + 16], %f4 363 fshiftorx %f0, %f2, %f10, %f0 364 fshiftorx %f2, %f4, %f10, %f2 365 366.L128aligned: 367 mov 10, $bits 368 and $inc, `10*16`, $tmp 369 st $bits, [$out + 240] ! store rounds 370 add $out, $tmp, $out ! start or end of key schedule 371 sllx $inc, 4, $inc ! 16 or -16 372___ 373for ($i=0; $i<10; $i++) { 374 $code.=<<___; 375 std %f0, [$out + 0] 376 faeskeyx %f2, `0x10+$i`, %f0 377 std %f2, [$out + 8] 378 add $out, $inc, $out 379 faeskeyx %f0, 0x00, %f2 380___ 381} 382$code.=<<___; 383 std %f0, [$out + 0] 384 std %f2, [$out + 8] 385 retl 386 xor %o0, %o0, %o0 ! return 0 387.type aes_fx_set_encrypt_key,#function 388.size aes_fx_set_encrypt_key,.-aes_fx_set_encrypt_key 389___ 390} 391{ 392my ($inp,$out,$len,$key,$ivp,$dir) = map("%i$_",(0..5)); 393my ($rounds,$inner,$end,$inc,$ialign,$oalign,$mask) = map("%l$_",(0..7)); 394my ($iv0,$iv1,$r0hi,$r0lo,$rlhi,$rllo,$in0,$in1,$intail,$outhead,$fshift) 395 = map("%f$_",grep { !($_ & 1) } (16 .. 62)); 396my ($ileft,$iright) = ($ialign,$oalign); 397 398$code.=<<___; 399.globl aes_fx_cbc_encrypt 400.align 32 401aes_fx_cbc_encrypt: 402 save %sp, -STACK_FRAME-16, %sp 403 srln $len, 4, $len 404 and $inp, 7, $ialign 405 andn $inp, 7, $inp 406 brz,pn $len, .Lcbc_no_data 407 sll $ialign, 3, $ileft 408 4091: call .+8 410 add %o7, .Linp_align-1b, %o7 411 412 ld [$key + 240], $rounds 413 and $out, 7, $oalign 414 ld [$ivp + 0], %f0 ! load ivec 415 andn $out, 7, $out 416 ld [$ivp + 4], %f1 417 sll $oalign, 3, $mask 418 ld [$ivp + 8], %f2 419 ld [$ivp + 12], %f3 420 421 sll $rounds, 4, $rounds 422 add $rounds, $key, $end 423 ldd [$key + 0], $r0hi ! round[0] 424 ldd [$key + 8], $r0lo 425 426 add $inp, 16, $inp 427 sub $len, 1, $len 428 ldd [$end + 0], $rlhi ! round[last] 429 ldd [$end + 8], $rllo 430 431 mov 16, $inc 432 movrz $len, 0, $inc 433 ldd [$key + 16], %f10 ! round[1] 434 ldd [$key + 24], %f12 435 436 ldd [%o7 + $ileft], $fshift ! shift left params 437 add %o7, 64, %o7 438 ldd [$inp - 16], $in0 ! load input 439 ldd [$inp - 8], $in1 440 ldda [$inp]0x82, $intail ! non-faulting load 441 brz $dir, .Lcbc_decrypt 442 add $inp, $inc, $inp ! inp+=16 443 444 fxor $r0hi, %f0, %f0 ! ivec^=round[0] 445 fxor $r0lo, %f2, %f2 446 fshiftorx $in0, $in1, $fshift, $in0 447 fshiftorx $in1, $intail, $fshift, $in1 448 nop 449 450.Loop_cbc_enc: 451 fxor $in0, %f0, %f0 ! inp^ivec^round[0] 452 fxor $in1, %f2, %f2 453 ldd [$key + 32], %f6 ! round[2] 454 ldd [$key + 40], %f8 455 add $key, 32, $end 456 sub $rounds, 16*6, $inner 457 458.Lcbc_enc: 459 fmovd %f0, %f4 460 faesencx %f2, %f10, %f0 461 faesencx %f4, %f12, %f2 462 ldd [$end + 16], %f10 463 ldd [$end + 24], %f12 464 add $end, 32, $end 465 466 fmovd %f0, %f4 467 faesencx %f2, %f6, %f0 468 faesencx %f4, %f8, %f2 469 ldd [$end + 0], %f6 470 ldd [$end + 8], %f8 471 472 brnz,a $inner, .Lcbc_enc 473 sub $inner, 16*2, $inner 474 475 fmovd %f0, %f4 476 faesencx %f2, %f10, %f0 477 faesencx %f4, %f12, %f2 478 ldd [$end + 16], %f10 ! round[last-1] 479 ldd [$end + 24], %f12 480 481 movrz $len, 0, $inc 482 483 brz,pn $len, .Lcbc_enc_skip_load 484 nop 485 486 fmovd $intail, $in0 487 ldd [$inp - 8], $in1 ! load next input block 488 ldda [$inp]0x82, $intail ! non-faulting load 489 add $inp, $inc, $inp ! inp+=16 490 491.Lcbc_enc_skip_load: 492 fmovd %f0, %f4 493 faesencx %f2, %f6, %f0 494 faesencx %f4, %f8, %f2 495 496 fshiftorx $in0, $in1, $fshift, $in0 497 fshiftorx $in1, $intail, $fshift, $in1 498 499 fmovd %f0, %f4 500 faesencx %f2, %f10, %f0 501 faesencx %f4, %f12, %f2 502 ldd [$key + 16], %f10 ! round[1] 503 ldd [$key + 24], %f12 504 505 fxor $r0hi, $in0, $in0 ! inp^=round[0] 506 fxor $r0lo, $in1, $in1 507 508 fmovd %f0, %f4 509 faesenclx %f2, $rlhi, %f0 510 faesenclx %f4, $rllo, %f2 511 512 brnz,pn $oalign, .Lcbc_enc_unaligned_out 513 nop 514 515 std %f0, [$out + 0] 516 std %f2, [$out + 8] 517 add $out, 16, $out 518 519 brnz,a $len, .Loop_cbc_enc 520 sub $len, 1, $len 521 522 st %f0, [$ivp + 0] ! output ivec 523 st %f1, [$ivp + 4] 524 st %f2, [$ivp + 8] 525 st %f3, [$ivp + 12] 526 527.Lcbc_no_data: 528 ret 529 restore 530 531.align 32 532.Lcbc_enc_unaligned_out: 533 ldd [%o7 + $mask], $fshift ! shift right params 534 mov 0xff, $mask 535 srl $mask, $oalign, $mask 536 sub %g0, $ileft, $iright 537 538 fshiftorx %f0, %f0, $fshift, %f6 539 fshiftorx %f0, %f2, $fshift, %f8 540 541 stda %f6, [$out + $mask]0xc0 ! partial store 542 orn %g0, $mask, $mask 543 std %f8, [$out + 8] 544 add $out, 16, $out 545 brz $len, .Lcbc_enc_unaligned_out_done 546 sub $len, 1, $len 547 b .Loop_cbc_enc_unaligned_out 548 nop 549 550.align 32 551.Loop_cbc_enc_unaligned_out: 552 fmovd %f2, $outhead 553 fxor $in0, %f0, %f0 ! inp^ivec^round[0] 554 fxor $in1, %f2, %f2 555 ldd [$key + 32], %f6 ! round[2] 556 ldd [$key + 40], %f8 557 558 fmovd %f0, %f4 559 faesencx %f2, %f10, %f0 560 faesencx %f4, %f12, %f2 561 ldd [$key + 48], %f10 ! round[3] 562 ldd [$key + 56], %f12 563 564 ldx [$inp - 16], %o0 565 ldx [$inp - 8], %o1 566 brz $ileft, .Lcbc_enc_aligned_inp 567 movrz $len, 0, $inc 568 569 ldx [$inp], %o2 570 sllx %o0, $ileft, %o0 571 srlx %o1, $iright, %g1 572 sllx %o1, $ileft, %o1 573 or %g1, %o0, %o0 574 srlx %o2, $iright, %o2 575 or %o2, %o1, %o1 576 577.Lcbc_enc_aligned_inp: 578 fmovd %f0, %f4 579 faesencx %f2, %f6, %f0 580 faesencx %f4, %f8, %f2 581 ldd [$key + 64], %f6 ! round[4] 582 ldd [$key + 72], %f8 583 add $key, 64, $end 584 sub $rounds, 16*8, $inner 585 586 stx %o0, [%sp + LOCALS + 0] 587 stx %o1, [%sp + LOCALS + 8] 588 add $inp, $inc, $inp ! inp+=16 589 nop 590 591.Lcbc_enc_unaligned: 592 fmovd %f0, %f4 593 faesencx %f2, %f10, %f0 594 faesencx %f4, %f12, %f2 595 ldd [$end + 16], %f10 596 ldd [$end + 24], %f12 597 add $end, 32, $end 598 599 fmovd %f0, %f4 600 faesencx %f2, %f6, %f0 601 faesencx %f4, %f8, %f2 602 ldd [$end + 0], %f6 603 ldd [$end + 8], %f8 604 605 brnz,a $inner, .Lcbc_enc_unaligned 606 sub $inner, 16*2, $inner 607 608 fmovd %f0, %f4 609 faesencx %f2, %f10, %f0 610 faesencx %f4, %f12, %f2 611 ldd [$end + 16], %f10 ! round[last-1] 612 ldd [$end + 24], %f12 613 614 fmovd %f0, %f4 615 faesencx %f2, %f6, %f0 616 faesencx %f4, %f8, %f2 617 618 ldd [%sp + LOCALS + 0], $in0 619 ldd [%sp + LOCALS + 8], $in1 620 621 fmovd %f0, %f4 622 faesencx %f2, %f10, %f0 623 faesencx %f4, %f12, %f2 624 ldd [$key + 16], %f10 ! round[1] 625 ldd [$key + 24], %f12 626 627 fxor $r0hi, $in0, $in0 ! inp^=round[0] 628 fxor $r0lo, $in1, $in1 629 630 fmovd %f0, %f4 631 faesenclx %f2, $rlhi, %f0 632 faesenclx %f4, $rllo, %f2 633 634 fshiftorx $outhead, %f0, $fshift, %f6 635 fshiftorx %f0, %f2, $fshift, %f8 636 std %f6, [$out + 0] 637 std %f8, [$out + 8] 638 add $out, 16, $out 639 640 brnz,a $len, .Loop_cbc_enc_unaligned_out 641 sub $len, 1, $len 642 643.Lcbc_enc_unaligned_out_done: 644 fshiftorx %f2, %f2, $fshift, %f8 645 stda %f8, [$out + $mask]0xc0 ! partial store 646 647 st %f0, [$ivp + 0] ! output ivec 648 st %f1, [$ivp + 4] 649 st %f2, [$ivp + 8] 650 st %f3, [$ivp + 12] 651 652 ret 653 restore 654 655.align 32 656.Lcbc_decrypt: 657 fshiftorx $in0, $in1, $fshift, $in0 658 fshiftorx $in1, $intail, $fshift, $in1 659 fmovd %f0, $iv0 660 fmovd %f2, $iv1 661 662.Loop_cbc_dec: 663 fxor $in0, $r0hi, %f0 ! inp^round[0] 664 fxor $in1, $r0lo, %f2 665 ldd [$key + 32], %f6 ! round[2] 666 ldd [$key + 40], %f8 667 add $key, 32, $end 668 sub $rounds, 16*6, $inner 669 670.Lcbc_dec: 671 fmovd %f0, %f4 672 faesdecx %f2, %f10, %f0 673 faesdecx %f4, %f12, %f2 674 ldd [$end + 16], %f10 675 ldd [$end + 24], %f12 676 add $end, 32, $end 677 678 fmovd %f0, %f4 679 faesdecx %f2, %f6, %f0 680 faesdecx %f4, %f8, %f2 681 ldd [$end + 0], %f6 682 ldd [$end + 8], %f8 683 684 brnz,a $inner, .Lcbc_dec 685 sub $inner, 16*2, $inner 686 687 fmovd %f0, %f4 688 faesdecx %f2, %f10, %f0 689 faesdecx %f4, %f12, %f2 690 ldd [$end + 16], %f10 ! round[last-1] 691 ldd [$end + 24], %f12 692 693 fmovd %f0, %f4 694 faesdecx %f2, %f6, %f0 695 faesdecx %f4, %f8, %f2 696 fxor $iv0, $rlhi, %f6 ! ivec^round[last] 697 fxor $iv1, $rllo, %f8 698 fmovd $in0, $iv0 699 fmovd $in1, $iv1 700 701 movrz $len, 0, $inc 702 703 brz,pn $len, .Lcbc_dec_skip_load 704 nop 705 706 fmovd $intail, $in0 707 ldd [$inp - 8], $in1 ! load next input block 708 ldda [$inp]0x82, $intail ! non-faulting load 709 add $inp, $inc, $inp ! inp+=16 710 711.Lcbc_dec_skip_load: 712 fmovd %f0, %f4 713 faesdecx %f2, %f10, %f0 714 faesdecx %f4, %f12, %f2 715 ldd [$key + 16], %f10 ! round[1] 716 ldd [$key + 24], %f12 717 718 fshiftorx $in0, $in1, $fshift, $in0 719 fshiftorx $in1, $intail, $fshift, $in1 720 721 fmovd %f0, %f4 722 faesdeclx %f2, %f6, %f0 723 faesdeclx %f4, %f8, %f2 724 725 brnz,pn $oalign, .Lcbc_dec_unaligned_out 726 nop 727 728 std %f0, [$out + 0] 729 std %f2, [$out + 8] 730 add $out, 16, $out 731 732 brnz,a $len, .Loop_cbc_dec 733 sub $len, 1, $len 734 735 st $iv0, [$ivp + 0] ! output ivec 736 st $iv0#lo, [$ivp + 4] 737 st $iv1, [$ivp + 8] 738 st $iv1#lo, [$ivp + 12] 739 740 ret 741 restore 742 743.align 32 744.Lcbc_dec_unaligned_out: 745 ldd [%o7 + $mask], $fshift ! shift right params 746 mov 0xff, $mask 747 srl $mask, $oalign, $mask 748 sub %g0, $ileft, $iright 749 750 fshiftorx %f0, %f0, $fshift, %f6 751 fshiftorx %f0, %f2, $fshift, %f8 752 753 stda %f6, [$out + $mask]0xc0 ! partial store 754 orn %g0, $mask, $mask 755 std %f8, [$out + 8] 756 add $out, 16, $out 757 brz $len, .Lcbc_dec_unaligned_out_done 758 sub $len, 1, $len 759 b .Loop_cbc_dec_unaligned_out 760 nop 761 762.align 32 763.Loop_cbc_dec_unaligned_out: 764 fmovd %f2, $outhead 765 fxor $in0, $r0hi, %f0 ! inp^round[0] 766 fxor $in1, $r0lo, %f2 767 ldd [$key + 32], %f6 ! round[2] 768 ldd [$key + 40], %f8 769 770 fmovd %f0, %f4 771 faesdecx %f2, %f10, %f0 772 faesdecx %f4, %f12, %f2 773 ldd [$key + 48], %f10 ! round[3] 774 ldd [$key + 56], %f12 775 776 ldx [$inp - 16], %o0 777 ldx [$inp - 8], %o1 778 brz $ileft, .Lcbc_dec_aligned_inp 779 movrz $len, 0, $inc 780 781 ldx [$inp], %o2 782 sllx %o0, $ileft, %o0 783 srlx %o1, $iright, %g1 784 sllx %o1, $ileft, %o1 785 or %g1, %o0, %o0 786 srlx %o2, $iright, %o2 787 or %o2, %o1, %o1 788 789.Lcbc_dec_aligned_inp: 790 fmovd %f0, %f4 791 faesdecx %f2, %f6, %f0 792 faesdecx %f4, %f8, %f2 793 ldd [$key + 64], %f6 ! round[4] 794 ldd [$key + 72], %f8 795 add $key, 64, $end 796 sub $rounds, 16*8, $inner 797 798 stx %o0, [%sp + LOCALS + 0] 799 stx %o1, [%sp + LOCALS + 8] 800 add $inp, $inc, $inp ! inp+=16 801 nop 802 803.Lcbc_dec_unaligned: 804 fmovd %f0, %f4 805 faesdecx %f2, %f10, %f0 806 faesdecx %f4, %f12, %f2 807 ldd [$end + 16], %f10 808 ldd [$end + 24], %f12 809 add $end, 32, $end 810 811 fmovd %f0, %f4 812 faesdecx %f2, %f6, %f0 813 faesdecx %f4, %f8, %f2 814 ldd [$end + 0], %f6 815 ldd [$end + 8], %f8 816 817 brnz,a $inner, .Lcbc_dec_unaligned 818 sub $inner, 16*2, $inner 819 820 fmovd %f0, %f4 821 faesdecx %f2, %f10, %f0 822 faesdecx %f4, %f12, %f2 823 ldd [$end + 16], %f10 ! round[last-1] 824 ldd [$end + 24], %f12 825 826 fmovd %f0, %f4 827 faesdecx %f2, %f6, %f0 828 faesdecx %f4, %f8, %f2 829 830 fxor $iv0, $rlhi, %f6 ! ivec^round[last] 831 fxor $iv1, $rllo, %f8 832 fmovd $in0, $iv0 833 fmovd $in1, $iv1 834 ldd [%sp + LOCALS + 0], $in0 835 ldd [%sp + LOCALS + 8], $in1 836 837 fmovd %f0, %f4 838 faesdecx %f2, %f10, %f0 839 faesdecx %f4, %f12, %f2 840 ldd [$key + 16], %f10 ! round[1] 841 ldd [$key + 24], %f12 842 843 fmovd %f0, %f4 844 faesdeclx %f2, %f6, %f0 845 faesdeclx %f4, %f8, %f2 846 847 fshiftorx $outhead, %f0, $fshift, %f6 848 fshiftorx %f0, %f2, $fshift, %f8 849 std %f6, [$out + 0] 850 std %f8, [$out + 8] 851 add $out, 16, $out 852 853 brnz,a $len, .Loop_cbc_dec_unaligned_out 854 sub $len, 1, $len 855 856.Lcbc_dec_unaligned_out_done: 857 fshiftorx %f2, %f2, $fshift, %f8 858 stda %f8, [$out + $mask]0xc0 ! partial store 859 860 st $iv0, [$ivp + 0] ! output ivec 861 st $iv0#lo, [$ivp + 4] 862 st $iv1, [$ivp + 8] 863 st $iv1#lo, [$ivp + 12] 864 865 ret 866 restore 867.type aes_fx_cbc_encrypt,#function 868.size aes_fx_cbc_encrypt,.-aes_fx_cbc_encrypt 869___ 870} 871{ 872my ($inp,$out,$len,$key,$ivp) = map("%i$_",(0..5)); 873my ($rounds,$inner,$end,$inc,$ialign,$oalign,$mask) = map("%l$_",(0..7)); 874my ($ctr0,$ctr1,$r0hi,$r0lo,$rlhi,$rllo,$in0,$in1,$intail,$outhead,$fshift) 875 = map("%f$_",grep { !($_ & 1) } (16 .. 62)); 876my ($ileft,$iright) = ($ialign, $oalign); 877my $one = "%f14"; 878 879$code.=<<___; 880.globl aes_fx_ctr32_encrypt_blocks 881.align 32 882aes_fx_ctr32_encrypt_blocks: 883 save %sp, -STACK_FRAME-16, %sp 884 srln $len, 0, $len 885 and $inp, 7, $ialign 886 andn $inp, 7, $inp 887 brz,pn $len, .Lctr32_no_data 888 sll $ialign, 3, $ileft 889 890.Lpic: call .+8 891 add %o7, .Linp_align - .Lpic, %o7 892 893 ld [$key + 240], $rounds 894 and $out, 7, $oalign 895 ld [$ivp + 0], $ctr0 ! load counter 896 andn $out, 7, $out 897 ld [$ivp + 4], $ctr0#lo 898 sll $oalign, 3, $mask 899 ld [$ivp + 8], $ctr1 900 ld [$ivp + 12], $ctr1#lo 901 ldd [%o7 + 128], $one 902 903 sll $rounds, 4, $rounds 904 add $rounds, $key, $end 905 ldd [$key + 0], $r0hi ! round[0] 906 ldd [$key + 8], $r0lo 907 908 add $inp, 16, $inp 909 sub $len, 1, $len 910 ldd [$key + 16], %f10 ! round[1] 911 ldd [$key + 24], %f12 912 913 mov 16, $inc 914 movrz $len, 0, $inc 915 ldd [$end + 0], $rlhi ! round[last] 916 ldd [$end + 8], $rllo 917 918 ldd [%o7 + $ileft], $fshift ! shiftleft params 919 add %o7, 64, %o7 920 ldd [$inp - 16], $in0 ! load input 921 ldd [$inp - 8], $in1 922 ldda [$inp]0x82, $intail ! non-faulting load 923 add $inp, $inc, $inp ! inp+=16 924 925 fshiftorx $in0, $in1, $fshift, $in0 926 fshiftorx $in1, $intail, $fshift, $in1 927 928.Loop_ctr32: 929 fxor $ctr0, $r0hi, %f0 ! counter^round[0] 930 fxor $ctr1, $r0lo, %f2 931 ldd [$key + 32], %f6 ! round[2] 932 ldd [$key + 40], %f8 933 add $key, 32, $end 934 sub $rounds, 16*6, $inner 935 936.Lctr32_enc: 937 fmovd %f0, %f4 938 faesencx %f2, %f10, %f0 939 faesencx %f4, %f12, %f2 940 ldd [$end + 16], %f10 941 ldd [$end + 24], %f12 942 add $end, 32, $end 943 944 fmovd %f0, %f4 945 faesencx %f2, %f6, %f0 946 faesencx %f4, %f8, %f2 947 ldd [$end + 0], %f6 948 ldd [$end + 8], %f8 949 950 brnz,a $inner, .Lctr32_enc 951 sub $inner, 16*2, $inner 952 953 fmovd %f0, %f4 954 faesencx %f2, %f10, %f0 955 faesencx %f4, %f12, %f2 956 ldd [$end + 16], %f10 ! round[last-1] 957 ldd [$end + 24], %f12 958 959 fmovd %f0, %f4 960 faesencx %f2, %f6, %f0 961 faesencx %f4, %f8, %f2 962 fxor $in0, $rlhi, %f6 ! inp^round[last] 963 fxor $in1, $rllo, %f8 964 965 movrz $len, 0, $inc 966 967 brz,pn $len, .Lctr32_enc_skip_load 968 nop 969 970 fmovd $intail, $in0 971 ldd [$inp - 8], $in1 ! load next input block 972 ldda [$inp]0x82, $intail ! non-faulting load 973 add $inp, $inc, $inp ! inp+=16 974 975.Lctr32_enc_skip_load: 976 fmovd %f0, %f4 977 faesencx %f2, %f10, %f0 978 faesencx %f4, %f12, %f2 979 ldd [$key + 16], %f10 ! round[1] 980 ldd [$key + 24], %f12 981 982 fshiftorx $in0, $in1, $fshift, $in0 983 fshiftorx $in1, $intail, $fshift, $in1 984 fpadd32 $ctr1, $one, $ctr1 ! increment counter 985 986 fmovd %f0, %f4 987 faesenclx %f2, %f6, %f0 988 faesenclx %f4, %f8, %f2 989 990 brnz,pn $oalign, .Lctr32_unaligned_out 991 nop 992 993 std %f0, [$out + 0] 994 std %f2, [$out + 8] 995 add $out, 16, $out 996 997 brnz,a $len, .Loop_ctr32 998 sub $len, 1, $len 999 1000.Lctr32_no_data: 1001 ret 1002 restore 1003 1004.align 32 1005.Lctr32_unaligned_out: 1006 ldd [%o7 + $mask], $fshift ! shift right params 1007 mov 0xff, $mask 1008 srl $mask, $oalign, $mask 1009 sub %g0, $ileft, $iright 1010 1011 fshiftorx %f0, %f0, $fshift, %f6 1012 fshiftorx %f0, %f2, $fshift, %f8 1013 1014 stda %f6, [$out + $mask]0xc0 ! partial store 1015 orn %g0, $mask, $mask 1016 std %f8, [$out + 8] 1017 add $out, 16, $out 1018 brz $len, .Lctr32_unaligned_out_done 1019 sub $len, 1, $len 1020 b .Loop_ctr32_unaligned_out 1021 nop 1022 1023.align 32 1024.Loop_ctr32_unaligned_out: 1025 fmovd %f2, $outhead 1026 fxor $ctr0, $r0hi, %f0 ! counter^round[0] 1027 fxor $ctr1, $r0lo, %f2 1028 ldd [$key + 32], %f6 ! round[2] 1029 ldd [$key + 40], %f8 1030 1031 fmovd %f0, %f4 1032 faesencx %f2, %f10, %f0 1033 faesencx %f4, %f12, %f2 1034 ldd [$key + 48], %f10 ! round[3] 1035 ldd [$key + 56], %f12 1036 1037 ldx [$inp - 16], %o0 1038 ldx [$inp - 8], %o1 1039 brz $ileft, .Lctr32_aligned_inp 1040 movrz $len, 0, $inc 1041 1042 ldx [$inp], %o2 1043 sllx %o0, $ileft, %o0 1044 srlx %o1, $iright, %g1 1045 sllx %o1, $ileft, %o1 1046 or %g1, %o0, %o0 1047 srlx %o2, $iright, %o2 1048 or %o2, %o1, %o1 1049 1050.Lctr32_aligned_inp: 1051 fmovd %f0, %f4 1052 faesencx %f2, %f6, %f0 1053 faesencx %f4, %f8, %f2 1054 ldd [$key + 64], %f6 ! round[4] 1055 ldd [$key + 72], %f8 1056 add $key, 64, $end 1057 sub $rounds, 16*8, $inner 1058 1059 stx %o0, [%sp + LOCALS + 0] 1060 stx %o1, [%sp + LOCALS + 8] 1061 add $inp, $inc, $inp ! inp+=16 1062 nop 1063 1064.Lctr32_enc_unaligned: 1065 fmovd %f0, %f4 1066 faesencx %f2, %f10, %f0 1067 faesencx %f4, %f12, %f2 1068 ldd [$end + 16], %f10 1069 ldd [$end + 24], %f12 1070 add $end, 32, $end 1071 1072 fmovd %f0, %f4 1073 faesencx %f2, %f6, %f0 1074 faesencx %f4, %f8, %f2 1075 ldd [$end + 0], %f6 1076 ldd [$end + 8], %f8 1077 1078 brnz,a $inner, .Lctr32_enc_unaligned 1079 sub $inner, 16*2, $inner 1080 1081 fmovd %f0, %f4 1082 faesencx %f2, %f10, %f0 1083 faesencx %f4, %f12, %f2 1084 ldd [$end + 16], %f10 ! round[last-1] 1085 ldd [$end + 24], %f12 1086 fpadd32 $ctr1, $one, $ctr1 ! increment counter 1087 1088 fmovd %f0, %f4 1089 faesencx %f2, %f6, %f0 1090 faesencx %f4, %f8, %f2 1091 fxor $in0, $rlhi, %f6 ! inp^round[last] 1092 fxor $in1, $rllo, %f8 1093 ldd [%sp + LOCALS + 0], $in0 1094 ldd [%sp + LOCALS + 8], $in1 1095 1096 fmovd %f0, %f4 1097 faesencx %f2, %f10, %f0 1098 faesencx %f4, %f12, %f2 1099 ldd [$key + 16], %f10 ! round[1] 1100 ldd [$key + 24], %f12 1101 1102 fmovd %f0, %f4 1103 faesenclx %f2, %f6, %f0 1104 faesenclx %f4, %f8, %f2 1105 1106 fshiftorx $outhead, %f0, $fshift, %f6 1107 fshiftorx %f0, %f2, $fshift, %f8 1108 std %f6, [$out + 0] 1109 std %f8, [$out + 8] 1110 add $out, 16, $out 1111 1112 brnz,a $len, .Loop_ctr32_unaligned_out 1113 sub $len, 1, $len 1114 1115.Lctr32_unaligned_out_done: 1116 fshiftorx %f2, %f2, $fshift, %f8 1117 stda %f8, [$out + $mask]0xc0 ! partial store 1118 1119 ret 1120 restore 1121.type aes_fx_ctr32_encrypt_blocks,#function 1122.size aes_fx_ctr32_encrypt_blocks,.-aes_fx_ctr32_encrypt_blocks 1123 1124.align 32 1125.Linp_align: ! fshiftorx parameters for left shift toward %rs1 1126 .byte 0, 0, 64, 0, 0, 64, 0, -64 1127 .byte 0, 0, 56, 8, 0, 56, 8, -56 1128 .byte 0, 0, 48, 16, 0, 48, 16, -48 1129 .byte 0, 0, 40, 24, 0, 40, 24, -40 1130 .byte 0, 0, 32, 32, 0, 32, 32, -32 1131 .byte 0, 0, 24, 40, 0, 24, 40, -24 1132 .byte 0, 0, 16, 48, 0, 16, 48, -16 1133 .byte 0, 0, 8, 56, 0, 8, 56, -8 1134.Lout_align: ! fshiftorx parameters for right shift toward %rs2 1135 .byte 0, 0, 0, 64, 0, 0, 64, 0 1136 .byte 0, 0, 8, 56, 0, 8, 56, -8 1137 .byte 0, 0, 16, 48, 0, 16, 48, -16 1138 .byte 0, 0, 24, 40, 0, 24, 40, -24 1139 .byte 0, 0, 32, 32, 0, 32, 32, -32 1140 .byte 0, 0, 40, 24, 0, 40, 24, -40 1141 .byte 0, 0, 48, 16, 0, 48, 16, -48 1142 .byte 0, 0, 56, 8, 0, 56, 8, -56 1143.Lone: 1144 .word 0, 1 1145.asciz "AES for Fujitsu SPARC64 X, CRYPTOGAMS by <appro\@openssl.org>" 1146.align 4 1147___ 1148} 1149# Purpose of these subroutines is to explicitly encode VIS instructions, 1150# so that one can compile the module without having to specify VIS 1151# extensions on compiler command line, e.g. -xarch=v9 vs. -xarch=v9a. 1152# Idea is to reserve for option to produce "universal" binary and let 1153# programmer detect if current CPU is VIS capable at run-time. 1154sub unvis { 1155my ($mnemonic,$rs1,$rs2,$rd)=@_; 1156my ($ref,$opf); 1157my %visopf = ( "faligndata" => 0x048, 1158 "bshuffle" => 0x04c, 1159 "fpadd32" => 0x052, 1160 "fxor" => 0x06c, 1161 "fsrc2" => 0x078 ); 1162 1163 $ref = "$mnemonic\t$rs1,$rs2,$rd"; 1164 1165 if ($opf=$visopf{$mnemonic}) { 1166 foreach ($rs1,$rs2,$rd) { 1167 return $ref if (!/%f([0-9]{1,2})/); 1168 $_=$1; 1169 if ($1>=32) { 1170 return $ref if ($1&1); 1171 # re-encode for upper double register addressing 1172 $_=($1|$1>>5)&31; 1173 } 1174 } 1175 1176 return sprintf ".word\t0x%08x !%s", 1177 0x81b00000|$rd<<25|$rs1<<14|$opf<<5|$rs2, 1178 $ref; 1179 } else { 1180 return $ref; 1181 } 1182} 1183 1184sub unvis3 { 1185my ($mnemonic,$rs1,$rs2,$rd)=@_; 1186my %bias = ( "g" => 0, "o" => 8, "l" => 16, "i" => 24 ); 1187my ($ref,$opf); 1188my %visopf = ( "alignaddr" => 0x018, 1189 "bmask" => 0x019, 1190 "alignaddrl" => 0x01a ); 1191 1192 $ref = "$mnemonic\t$rs1,$rs2,$rd"; 1193 1194 if ($opf=$visopf{$mnemonic}) { 1195 foreach ($rs1,$rs2,$rd) { 1196 return $ref if (!/%([goli])([0-9])/); 1197 $_=$bias{$1}+$2; 1198 } 1199 1200 return sprintf ".word\t0x%08x !%s", 1201 0x81b00000|$rd<<25|$rs1<<14|$opf<<5|$rs2, 1202 $ref; 1203 } else { 1204 return $ref; 1205 } 1206} 1207 1208sub unfx { 1209my ($mnemonic,$rs1,$rs2,$rd)=@_; 1210my ($ref,$opf); 1211my %aesopf = ( "faesencx" => 0x90, 1212 "faesdecx" => 0x91, 1213 "faesenclx" => 0x92, 1214 "faesdeclx" => 0x93, 1215 "faeskeyx" => 0x94 ); 1216 1217 $ref = "$mnemonic\t$rs1,$rs2,$rd"; 1218 1219 if (defined($opf=$aesopf{$mnemonic})) { 1220 $rs2 = ($rs2 =~ /%f([0-6]*[02468])/) ? (($1|$1>>5)&31) : $rs2; 1221 $rs2 = oct($rs2) if ($rs2 =~ /^0/); 1222 1223 foreach ($rs1,$rd) { 1224 return $ref if (!/%f([0-9]{1,2})/); 1225 $_=$1; 1226 if ($1>=32) { 1227 return $ref if ($1&1); 1228 # re-encode for upper double register addressing 1229 $_=($1|$1>>5)&31; 1230 } 1231 } 1232 1233 return sprintf ".word\t0x%08x !%s", 1234 2<<30|$rd<<25|0x36<<19|$rs1<<14|$opf<<5|$rs2, 1235 $ref; 1236 } else { 1237 return $ref; 1238 } 1239} 1240 1241sub unfx3src { 1242my ($mnemonic,$rs1,$rs2,$rs3,$rd)=@_; 1243my ($ref,$opf); 1244my %aesopf = ( "fshiftorx" => 0x0b ); 1245 1246 $ref = "$mnemonic\t$rs1,$rs2,$rs3,$rd"; 1247 1248 if (defined($opf=$aesopf{$mnemonic})) { 1249 foreach ($rs1,$rs2,$rs3,$rd) { 1250 return $ref if (!/%f([0-9]{1,2})/); 1251 $_=$1; 1252 if ($1>=32) { 1253 return $ref if ($1&1); 1254 # re-encode for upper double register addressing 1255 $_=($1|$1>>5)&31; 1256 } 1257 } 1258 1259 return sprintf ".word\t0x%08x !%s", 1260 2<<30|$rd<<25|0x37<<19|$rs1<<14|$rs3<<9|$opf<<5|$rs2, 1261 $ref; 1262 } else { 1263 return $ref; 1264 } 1265} 1266 1267foreach (split("\n",$code)) { 1268 s/\`([^\`]*)\`/eval $1/ge; 1269 1270 s/%f([0-9]+)#lo/sprintf "%%f%d",$1+1/ge; 1271 1272 s/\b(faes[^x]{3,4}x)\s+(%f[0-9]{1,2}),\s*([%fx0-9]+),\s*(%f[0-9]{1,2})/ 1273 &unfx($1,$2,$3,$4) 1274 /ge or 1275 s/\b([f][^\s]*)\s+(%f[0-9]{1,2}),\s*(%f[0-9]{1,2}),\s*(%f[0-9]{1,2}),\s*(%f[0-9]{1,2})/ 1276 &unfx3src($1,$2,$3,$4,$5) 1277 /ge or 1278 s/\b([fb][^\s]*)\s+(%f[0-9]{1,2}),\s*(%f[0-9]{1,2}),\s*(%f[0-9]{1,2})/ 1279 &unvis($1,$2,$3,$4) 1280 /ge or 1281 s/\b(alignaddr[l]*)\s+(%[goli][0-7]),\s*(%[goli][0-7]),\s*(%[goli][0-7])/ 1282 &unvis3($1,$2,$3,$4) 1283 /ge; 1284 print $_,"\n"; 1285} 1286 1287close STDOUT or die "error closing STDOUT: $!"; 1288