e_padlock-x86_64.pl: brown-bag bug in stack pointer handling.
[openssl.git] / engines / asm / e_padlock-x86.pl
index df8f56b5214d7d2b735021443ddd0942936d1bc5..e211706ae1b8511e3db929c8ff59eaf6f1513757 100644 (file)
@@ -183,7 +183,7 @@ my ($mode,$opcode) = @_;
 &set_label("${mode}_pic_point");
        &lea    ($ctx,&DWP(16,$ctx));   # control word
        &xor    ("eax","eax");
-                                       if ($mode eq "ctr16") {
+                                       if ($mode eq "ctr32") {
        &movq   ("mm0",&QWP(-16,$ctx)); # load [upper part of] counter
                                        } else {
        &xor    ("ebx","ebx");
@@ -216,7 +216,7 @@ my ($mode,$opcode) = @_;
        &mov    (&DWP(8,"ebp"),$len);
        &mov    ($len,$chunk);
        &mov    (&DWP(12,"ebp"),$chunk);        # chunk
-                                               if ($mode eq "ctr16") {
+                                               if ($mode eq "ctr32") {
        &mov    ("ecx",&DWP(-4,$ctx));
        &xor    ($out,$out);
        &mov    ("eax",&DWP(-8,$ctx));          # borrow $len
@@ -257,7 +257,7 @@ my ($mode,$opcode) = @_;
                                                }
        &mov    ($out,&DWP(0,"ebp"));           # restore parameters
        &mov    ($chunk,&DWP(12,"ebp"));
-                                               if ($mode eq "ctr16") {
+                                               if ($mode eq "ctr32") {
        &mov    ($inp,&DWP(4,"ebp"));
        &xor    ($len,$len);
 &set_label("${mode}_xor");
@@ -284,7 +284,7 @@ my ($mode,$opcode) = @_;
        &sub    ($len,$chunk);
        &mov    ($chunk,$PADLOCK_CHUNK);
        &jnz    (&label("${mode}_loop"));
-                                               if ($mode ne "ctr16") {
+                                               if ($mode ne "ctr32") {
        &test   ($out,0x0f);                    # out_misaligned
        &jz     (&label("${mode}_done"));
                                                }
@@ -296,7 +296,7 @@ my ($mode,$opcode) = @_;
        &data_byte(0xf3,0xab);                  # rep stosl
 &set_label("${mode}_done");
        &lea    ("esp",&DWP(24,"ebp"));
-                                               if ($mode ne "ctr16") {
+                                               if ($mode ne "ctr32") {
        &jmp    (&label("${mode}_exit"));
 
 &set_label("${mode}_aligned",16);
@@ -311,7 +311,7 @@ my ($mode,$opcode) = @_;
 &set_label("${mode}_exit");                    }
        &mov    ("eax",1);
        &lea    ("esp",&DWP(4,"esp"));          # popf
-       &emms   ()                              if ($mode eq "ctr16");
+       &emms   ()                              if ($mode eq "ctr32");
 &set_label("${mode}_abort");
 &function_end("padlock_${mode}_encrypt");
 }
@@ -320,10 +320,11 @@ my ($mode,$opcode) = @_;
 &generate_mode("cbc",0xd0);
 &generate_mode("cfb",0xe0);
 &generate_mode("ofb",0xe8);
-&generate_mode("ctr16",0xc8);  # yes, it implements own ctr with ecb opcode,
-                               # because hardware ctr was introduced later
-                               # and even has errata on certain CPU stepping.
-                               # own implementation *always* works...
+&generate_mode("ctr32",0xc8);  # yes, it implements own CTR with ECB opcode,
+                               # because hardware CTR was introduced later
+                               # and even has errata on certain C7 stepping.
+                               # own implementation *always* works, though
+                               # ~15% slower than dedicated hardware...
 
 &function_begin_B("padlock_xstore");
        &push   ("edi");
@@ -351,19 +352,34 @@ my ($mode,$opcode) = @_;
        &push   ("edi");
        &push   ("esi");
        &xor    ("eax","eax");
+       &mov    ("edi",&wparam(0));
+       &mov    ("esi",&wparam(1));
+       &mov    ("ecx",&wparam(2));
     if ($::win32 or $::coff) {
        &push   (&::islabel("_win32_segv_handler"));
        &data_byte(0x64,0xff,0x30);             # push  %fs:(%eax)
        &data_byte(0x64,0x89,0x20);             # mov   %esp,%fs:(%eax)
     }
-       &mov    ("edi",&wparam(0));
-       &mov    ("esi",&wparam(1));
-       &mov    ("ecx",&wparam(2));
+       &mov    ("edx","esp");                  # put aside %esp
+       &add    ("esp",-128);                   # 32 is enough but spec says 128
+       &movups ("xmm0",&QWP(0,"edi"));         # copy-in context
+       &and    ("esp",-16);
+       &mov    ("eax",&DWP(16,"edi"));
+       &movaps (&QWP(0,"esp"),"xmm0");
+       &mov    ("edi","esp");
+       &mov    (&DWP(16,"esp"),"eax");
+       &xor    ("eax","eax");
        &data_byte(0xf3,0x0f,0xa6,0xc8);        # rep xsha1
+       &movaps ("xmm0",&QWP(0,"esp"));
+       &mov    ("eax",&DWP(16,"esp"));
+       &mov    ("esp","edx");                  # restore %esp
     if ($::win32 or $::coff) {
        &data_byte(0x64,0x8f,0x05,0,0,0,0);     # pop   %fs:0
        &lea    ("esp",&DWP(4,"esp"));
     }
+       &mov    ("edi",&wparam(0));
+       &movups (&QWP(0,"edi"),"xmm0");         # copy-out context
+       &mov    (&DWP(16,"edi"),"eax");
        &pop    ("esi");
        &pop    ("edi");
        &ret    ();
@@ -372,12 +388,26 @@ my ($mode,$opcode) = @_;
 &function_begin_B("padlock_sha1_blocks");
        &push   ("edi");
        &push   ("esi");
-       &mov    ("eax",-1);
        &mov    ("edi",&wparam(0));
        &mov    ("esi",&wparam(1));
+       &mov    ("edx","esp");                  # put aside %esp
        &mov    ("ecx",&wparam(2));
+       &add    ("esp",-128);
+       &movups ("xmm0",&QWP(0,"edi"));         # copy-in context
+       &and    ("esp",-16);
+       &mov    ("eax",&DWP(16,"edi"));
+       &movaps (&QWP(0,"esp"),"xmm0");
+       &mov    ("edi","esp");
+       &mov    (&DWP(16,"esp"),"eax");
+       &mov    ("eax",-1);
        &data_byte(0xf3,0x0f,0xa6,0xc8);        # rep xsha1
-       &pop    ("esi");
+       &movaps ("xmm0",&QWP(0,"esp"));
+       &mov    ("eax",&DWP(16,"esp"));
+       &mov    ("esp","edx");                  # restore %esp
+       &mov    ("edi",&wparam(0));
+       &movups (&QWP(0,"edi"),"xmm0");         # copy-out context
+       &mov    (&DWP(16,"edi"),"eax");
+       &pop    ("esi");
        &pop    ("edi");
        &ret    ();
 &function_end_B("padlock_sha1_blocks");
@@ -386,19 +416,34 @@ my ($mode,$opcode) = @_;
        &push   ("edi");
        &push   ("esi");
        &xor    ("eax","eax");
+       &mov    ("edi",&wparam(0));
+       &mov    ("esi",&wparam(1));
+       &mov    ("ecx",&wparam(2));
     if ($::win32 or $::coff) {
        &push   (&::islabel("_win32_segv_handler"));
        &data_byte(0x64,0xff,0x30);             # push  %fs:(%eax)
        &data_byte(0x64,0x89,0x20);             # mov   %esp,%fs:(%eax)
     }
-       &mov    ("edi",&wparam(0));
-       &mov    ("esi",&wparam(1));
-       &mov    ("ecx",&wparam(2));
+       &mov    ("edx","esp");                  # put aside %esp
+       &add    ("esp",-128);
+       &movups ("xmm0",&QWP(0,"edi"));         # copy-in context
+       &and    ("esp",-16);
+       &movups ("xmm1",&QWP(16,"edi"));
+       &movaps (&QWP(0,"esp"),"xmm0");
+       &mov    ("edi","esp");
+       &movaps (&QWP(16,"esp"),"xmm1");
+       &xor    ("eax","eax");
        &data_byte(0xf3,0x0f,0xa6,0xd0);        # rep xsha256
+       &movaps ("xmm0",&QWP(0,"esp"));
+       &movaps ("xmm1",&QWP(16,"esp"));
+       &mov    ("esp","edx");                  # restore %esp
     if ($::win32 or $::coff) {
        &data_byte(0x64,0x8f,0x05,0,0,0,0);     # pop   %fs:0
        &lea    ("esp",&DWP(4,"esp"));
     }
+       &mov    ("edi",&wparam(0));
+       &movups (&QWP(0,"edi"),"xmm0");         # copy-out context
+       &movups (&QWP(16,"edi"),"xmm1");
        &pop    ("esi");
        &pop    ("edi");
        &ret    ();
@@ -407,11 +452,25 @@ my ($mode,$opcode) = @_;
 &function_begin_B("padlock_sha256_blocks");
        &push   ("edi");
        &push   ("esi");
-       &mov    ("eax",-1);
        &mov    ("edi",&wparam(0));
        &mov    ("esi",&wparam(1));
        &mov    ("ecx",&wparam(2));
+       &mov    ("edx","esp");                  # put aside %esp
+       &add    ("esp",-128);
+       &movups ("xmm0",&QWP(0,"edi"));         # copy-in context
+       &and    ("esp",-16);
+       &movups ("xmm1",&QWP(16,"edi"));
+       &movaps (&QWP(0,"esp"),"xmm0");
+       &mov    ("edi","esp");
+       &movaps (&QWP(16,"esp"),"xmm1");
+       &mov    ("eax",-1);
        &data_byte(0xf3,0x0f,0xa6,0xd0);        # rep xsha256
+       &movaps ("xmm0",&QWP(0,"esp"));
+       &movaps ("xmm1",&QWP(16,"esp"));
+       &mov    ("esp","edx");                  # restore %esp
+       &mov    ("edi",&wparam(0));
+       &movups (&QWP(0,"edi"),"xmm0");         # copy-out context
+       &movups (&QWP(16,"edi"),"xmm1");
        &pop    ("esi");
        &pop    ("edi");
        &ret    ();
@@ -423,7 +482,29 @@ my ($mode,$opcode) = @_;
        &mov    ("edi",&wparam(0));
        &mov    ("esi",&wparam(1));
        &mov    ("ecx",&wparam(2));
+       &mov    ("edx","esp");                  # put aside %esp
+       &add    ("esp",-128);
+       &movups ("xmm0",&QWP(0,"edi"));         # copy-in context
+       &and    ("esp",-16);
+       &movups ("xmm1",&QWP(16,"edi"));
+       &movups ("xmm2",&QWP(32,"edi"));
+       &movups ("xmm3",&QWP(48,"edi"));
+       &movaps (&QWP(0,"esp"),"xmm0");
+       &mov    ("edi","esp");
+       &movaps (&QWP(16,"esp"),"xmm1");
+       &movaps (&QWP(32,"esp"),"xmm2");
+       &movaps (&QWP(48,"esp"),"xmm3");
        &data_byte(0xf3,0x0f,0xa6,0xe0);        # rep xsha512
+       &movaps ("xmm0",&QWP(0,"esp"));
+       &movaps ("xmm1",&QWP(16,"esp"));
+       &movaps ("xmm2",&QWP(32,"esp"));
+       &movaps ("xmm3",&QWP(48,"esp"));
+       &mov    ("esp","edx");                  # restore %esp
+       &mov    ("edi",&wparam(0));
+       &movups (&QWP(0,"edi"),"xmm0");         # copy-out context
+       &movups (&QWP(16,"edi"),"xmm1");
+       &movups (&QWP(32,"edi"),"xmm2");
+       &movups (&QWP(48,"edi"),"xmm3");
        &pop    ("esi");
        &pop    ("edi");
        &ret    ();