1 /*
   2  * Copyright (c) 2003, 2026, Oracle and/or its affiliates. All rights reserved.
   3  * Copyright (c) 2014, 2025, Red Hat Inc. All rights reserved.
   4  * Copyright (c) 2020, 2025, Huawei Technologies Co., Ltd. All rights reserved.
   5  * DO NOT ALTER OR REMOVE COPYRIGHT NOTICES OR THIS FILE HEADER.
   6  *
   7  * This code is free software; you can redistribute it and/or modify it
   8  * under the terms of the GNU General Public License version 2 only, as
   9  * published by the Free Software Foundation.
  10  *
  11  * This code is distributed in the hope that it will be useful, but WITHOUT
  12  * ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
  13  * FITNESS FOR A PARTICULAR PURPOSE.  See the GNU General Public License
  14  * version 2 for more details (a copy is included in the LICENSE file that
  15  * accompanied this code).
  16  *
  17  * You should have received a copy of the GNU General Public License version
  18  * 2 along with this work; if not, write to the Free Software Foundation,
  19  * Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA.
  20  *
  21  * Please contact Oracle, 500 Oracle Parkway, Redwood Shores, CA 94065 USA
  22  * or visit www.oracle.com if you need additional information or have any
  23  * questions.
  24  *
  25  */
  26 
  27 #include "asm/macroAssembler.hpp"
  28 #include "asm/macroAssembler.inline.hpp"
  29 #include "compiler/oopMap.hpp"
  30 #include "gc/shared/barrierSet.hpp"
  31 #include "gc/shared/barrierSetAssembler.hpp"
  32 #include "interpreter/interpreter.hpp"
  33 #include "memory/universe.hpp"
  34 #include "nativeInst_riscv.hpp"
  35 #include "oops/instanceOop.hpp"
  36 #include "oops/method.hpp"
  37 #include "oops/objArrayKlass.hpp"
  38 #include "oops/oop.inline.hpp"
  39 #include "prims/methodHandles.hpp"
  40 #include "prims/upcallLinker.hpp"
  41 #include "runtime/continuation.hpp"
  42 #include "runtime/continuationEntry.inline.hpp"
  43 #include "runtime/frame.inline.hpp"
  44 #include "runtime/handles.inline.hpp"
  45 #include "runtime/javaThread.hpp"
  46 #include "runtime/sharedRuntime.hpp"
  47 #include "runtime/stubCodeGenerator.hpp"
  48 #include "runtime/stubRoutines.hpp"
  49 #include "utilities/align.hpp"
  50 #include "utilities/powerOfTwo.hpp"
  51 #ifdef COMPILER2
  52 #include "opto/runtime.hpp"
  53 #endif
  54 
  55 // Declaration and definition of StubGenerator (no .hpp file).
  56 // For a more detailed description of the stub routine structure
  57 // see the comment in stubRoutines.hpp
  58 
  59 #undef __
  60 #define __ _masm->
  61 
  62 #ifdef PRODUCT
  63 #define BLOCK_COMMENT(str) /* nothing */
  64 #else
  65 #define BLOCK_COMMENT(str) __ block_comment(str)
  66 #endif
  67 
  68 #define BIND(label) bind(label); BLOCK_COMMENT(#label ":")
  69 
  70 // Stub Code definitions
  71 
  72 class StubGenerator: public StubCodeGenerator {
  73  private:
  74 
  75 #ifdef PRODUCT
  76 #define inc_counter_np(counter) ((void)0)
  77 #else
  78   void inc_counter_np_(uint& counter) {
  79     __ incrementw(ExternalAddress((address)&counter));
  80   }
  81 #define inc_counter_np(counter) \
  82   BLOCK_COMMENT("inc_counter " #counter); \
  83   inc_counter_np_(counter);
  84 #endif
  85 
  86   // Call stubs are used to call Java from C
  87   //
  88   // Arguments:
  89   //    c_rarg0:   call wrapper address                   address
  90   //    c_rarg1:   result                                 address
  91   //    c_rarg2:   result type                            BasicType
  92   //    c_rarg3:   method                                 Method*
  93   //    c_rarg4:   (interpreter) entry point              address
  94   //    c_rarg5:   parameters                             intptr_t*
  95   //    c_rarg6:   parameter size (in words)              int
  96   //    c_rarg7:   thread                                 Thread*
  97   //
  98   // There is no return from the stub itself as any Java result
  99   // is written to result
 100   //
 101   // we save x1 (ra) as the return PC at the base of the frame and
 102   // link x8 (fp) below it as the frame pointer installing sp (x2)
 103   // into fp.
 104   //
 105   // we save x10-x17, which accounts for all the c arguments.
 106   //
 107   // TODO: strictly do we need to save them all? they are treated as
 108   // volatile by C so could we omit saving the ones we are going to
 109   // place in global registers (thread? method?) or those we only use
 110   // during setup of the Java call?
 111   //
 112   // we don't need to save x5 which C uses as an indirect result location
 113   // return register.
 114   //
 115   // we don't need to save x6-x7 and x28-x31 which both C and Java treat as
 116   // volatile
 117   //
 118   // we save x9, x18-x27, f8-f9, and f18-f27 which Java uses as temporary
 119   // registers and C expects to be callee-save
 120   //
 121   // so the stub frame looks like this when we enter Java code
 122   //
 123   //     [ return_from_Java     ] <--- sp
 124   //     [ argument word n      ]
 125   //      ...
 126   // -35 [ argument word 1      ]
 127   // -34 [ saved FRM in Floating-point Control and Status Register ] <--- sp_after_call
 128   // -33 [ saved f27            ]
 129   // -32 [ saved f26            ]
 130   // -31 [ saved f25            ]
 131   // -30 [ saved f24            ]
 132   // -29 [ saved f23            ]
 133   // -28 [ saved f22            ]
 134   // -27 [ saved f21            ]
 135   // -26 [ saved f20            ]
 136   // -25 [ saved f19            ]
 137   // -24 [ saved f18            ]
 138   // -23 [ saved f9             ]
 139   // -22 [ saved f8             ]
 140   // -21 [ saved x27            ]
 141   // -20 [ saved x26            ]
 142   // -19 [ saved x25            ]
 143   // -18 [ saved x24            ]
 144   // -17 [ saved x23            ]
 145   // -16 [ saved x22            ]
 146   // -15 [ saved x21            ]
 147   // -14 [ saved x20            ]
 148   // -13 [ saved x19            ]
 149   // -12 [ saved x18            ]
 150   // -11 [ saved x9             ]
 151   // -10 [ call wrapper   (x10) ]
 152   //  -9 [ result         (x11) ]
 153   //  -8 [ result type    (x12) ]
 154   //  -7 [ method         (x13) ]
 155   //  -6 [ entry point    (x14) ]
 156   //  -5 [ parameters     (x15) ]
 157   //  -4 [ parameter size (x16) ]
 158   //  -3 [ thread         (x17) ]
 159   //  -2 [ saved fp       (x8)  ]
 160   //  -1 [ saved ra       (x1)  ]
 161   //   0 [                      ] <--- fp == saved sp (x2)
 162 
 163   // Call stub stack layout word offsets from fp
 164   enum call_stub_layout {
 165     sp_after_call_off  = -34,
 166 
 167     frm_off            = sp_after_call_off,
 168     f27_off            = -33,
 169     f26_off            = -32,
 170     f25_off            = -31,
 171     f24_off            = -30,
 172     f23_off            = -29,
 173     f22_off            = -28,
 174     f21_off            = -27,
 175     f20_off            = -26,
 176     f19_off            = -25,
 177     f18_off            = -24,
 178     f9_off             = -23,
 179     f8_off             = -22,
 180 
 181     x27_off            = -21,
 182     x26_off            = -20,
 183     x25_off            = -19,
 184     x24_off            = -18,
 185     x23_off            = -17,
 186     x22_off            = -16,
 187     x21_off            = -15,
 188     x20_off            = -14,
 189     x19_off            = -13,
 190     x18_off            = -12,
 191     x9_off             = -11,
 192 
 193     call_wrapper_off   = -10,
 194     result_off         = -9,
 195     result_type_off    = -8,
 196     method_off         = -7,
 197     entry_point_off    = -6,
 198     parameters_off     = -5,
 199     parameter_size_off = -4,
 200     thread_off         = -3,
 201     fp_f               = -2,
 202     retaddr_off        = -1,
 203   };
 204 
 205   address generate_call_stub(address& return_address) {
 206     assert((int)frame::entry_frame_after_call_words == -(int)sp_after_call_off + 1 &&
 207            (int)frame::entry_frame_call_wrapper_offset == (int)call_wrapper_off,
 208            "adjust this code");
 209 
 210     StubId stub_id = StubId::stubgen_call_stub_id;
 211     StubCodeMark mark(this, stub_id);
 212     address start = __ pc();
 213 
 214     const Address sp_after_call (fp, sp_after_call_off  * wordSize);
 215 
 216     const Address frm_save      (fp, frm_off           * wordSize);
 217     const Address call_wrapper  (fp, call_wrapper_off   * wordSize);
 218     const Address result        (fp, result_off         * wordSize);
 219     const Address result_type   (fp, result_type_off    * wordSize);
 220     const Address method        (fp, method_off         * wordSize);
 221     const Address entry_point   (fp, entry_point_off    * wordSize);
 222     const Address parameters    (fp, parameters_off     * wordSize);
 223     const Address parameter_size(fp, parameter_size_off * wordSize);
 224 
 225     const Address thread        (fp, thread_off         * wordSize);
 226 
 227     const Address f27_save      (fp, f27_off            * wordSize);
 228     const Address f26_save      (fp, f26_off            * wordSize);
 229     const Address f25_save      (fp, f25_off            * wordSize);
 230     const Address f24_save      (fp, f24_off            * wordSize);
 231     const Address f23_save      (fp, f23_off            * wordSize);
 232     const Address f22_save      (fp, f22_off            * wordSize);
 233     const Address f21_save      (fp, f21_off            * wordSize);
 234     const Address f20_save      (fp, f20_off            * wordSize);
 235     const Address f19_save      (fp, f19_off            * wordSize);
 236     const Address f18_save      (fp, f18_off            * wordSize);
 237     const Address f9_save       (fp, f9_off             * wordSize);
 238     const Address f8_save       (fp, f8_off             * wordSize);
 239 
 240     const Address x27_save      (fp, x27_off            * wordSize);
 241     const Address x26_save      (fp, x26_off            * wordSize);
 242     const Address x25_save      (fp, x25_off            * wordSize);
 243     const Address x24_save      (fp, x24_off            * wordSize);
 244     const Address x23_save      (fp, x23_off            * wordSize);
 245     const Address x22_save      (fp, x22_off            * wordSize);
 246     const Address x21_save      (fp, x21_off            * wordSize);
 247     const Address x20_save      (fp, x20_off            * wordSize);
 248     const Address x19_save      (fp, x19_off            * wordSize);
 249     const Address x18_save      (fp, x18_off            * wordSize);
 250 
 251     const Address x9_save       (fp, x9_off             * wordSize);
 252 
 253     // stub code
 254 
 255     address riscv_entry = __ pc();
 256 
 257     // set up frame and move sp to end of save area
 258     __ enter();
 259     __ addi(sp, fp, sp_after_call_off * wordSize);
 260 
 261     // save register parameters and Java temporary/global registers
 262     // n.b. we save thread even though it gets installed in
 263     // xthread because we want to sanity check tp later
 264     __ sd(c_rarg7, thread);
 265     __ sw(c_rarg6, parameter_size);
 266     __ sd(c_rarg5, parameters);
 267     __ sd(c_rarg4, entry_point);
 268     __ sd(c_rarg3, method);
 269     __ sd(c_rarg2, result_type);
 270     __ sd(c_rarg1, result);
 271     __ sd(c_rarg0, call_wrapper);
 272 
 273     __ sd(x9, x9_save);
 274 
 275     __ sd(x18, x18_save);
 276     __ sd(x19, x19_save);
 277     __ sd(x20, x20_save);
 278     __ sd(x21, x21_save);
 279     __ sd(x22, x22_save);
 280     __ sd(x23, x23_save);
 281     __ sd(x24, x24_save);
 282     __ sd(x25, x25_save);
 283     __ sd(x26, x26_save);
 284     __ sd(x27, x27_save);
 285 
 286     __ fsd(f8,  f8_save);
 287     __ fsd(f9,  f9_save);
 288     __ fsd(f18, f18_save);
 289     __ fsd(f19, f19_save);
 290     __ fsd(f20, f20_save);
 291     __ fsd(f21, f21_save);
 292     __ fsd(f22, f22_save);
 293     __ fsd(f23, f23_save);
 294     __ fsd(f24, f24_save);
 295     __ fsd(f25, f25_save);
 296     __ fsd(f26, f26_save);
 297     __ fsd(f27, f27_save);
 298 
 299     __ frrm(t0);
 300     __ sd(t0, frm_save);
 301     // Set frm to the state we need. We do want Round to Nearest. We
 302     // don't want non-IEEE rounding modes.
 303     Label skip_fsrmi;
 304     guarantee(__ RoundingMode::rne == 0, "must be");
 305     __ beqz(t0, skip_fsrmi);
 306     __ fsrmi(__ RoundingMode::rne);
 307     __ bind(skip_fsrmi);
 308 
 309     // install Java thread in global register now we have saved
 310     // whatever value it held
 311     __ mv(xthread, c_rarg7);
 312 
 313     // And method
 314     __ mv(xmethod, c_rarg3);
 315 
 316     // set up the heapbase register
 317     __ reinit_heapbase();
 318 
 319 #ifdef ASSERT
 320     // make sure we have no pending exceptions
 321     {
 322       Label L;
 323       __ ld(t0, Address(xthread, in_bytes(Thread::pending_exception_offset())));
 324       __ beqz(t0, L);
 325       __ stop("StubRoutines::call_stub: entered with pending exception");
 326       __ BIND(L);
 327     }
 328 #endif
 329     // pass parameters if any
 330     __ mv(esp, sp);
 331     __ slli(t0, c_rarg6, LogBytesPerWord);
 332     __ sub(t0, sp, t0); // Move SP out of the way
 333     __ andi(sp, t0, -2 * wordSize);
 334 
 335     BLOCK_COMMENT("pass parameters if any");
 336     Label parameters_done;
 337     // parameter count is still in c_rarg6
 338     // and parameter pointer identifying param 1 is in c_rarg5
 339     __ beqz(c_rarg6, parameters_done);
 340 
 341     address loop = __ pc();
 342     __ ld(t0, Address(c_rarg5, 0));
 343     __ addi(c_rarg5, c_rarg5, wordSize);
 344     __ subi(c_rarg6, c_rarg6, 1);
 345     __ push_reg(t0);
 346     __ bgtz(c_rarg6, loop);
 347 
 348     __ BIND(parameters_done);
 349 
 350     // call Java entry -- passing methdoOop, and current sp
 351     //      xmethod: Method*
 352     //      x19_sender_sp: sender sp
 353     BLOCK_COMMENT("call Java function");
 354     __ mv(x19_sender_sp, sp);
 355     __ jalr(c_rarg4);
 356 
 357     // save current address for use by exception handling code
 358 
 359     return_address = __ pc();
 360 
 361     // store result depending on type (everything that is not
 362     // T_OBJECT, T_LONG, T_FLOAT or T_DOUBLE is treated as T_INT)
 363     // n.b. this assumes Java returns an integral result in x10
 364     // and a floating result in j_farg0
 365     __ ld(j_rarg2, result);
 366     Label is_long, is_float, is_double, exit;
 367     __ ld(j_rarg1, result_type);
 368     __ mv(t0, (u1)T_OBJECT);
 369     __ beq(j_rarg1, t0, is_long);
 370     __ mv(t0, (u1)T_LONG);
 371     __ beq(j_rarg1, t0, is_long);
 372     __ mv(t0, (u1)T_FLOAT);
 373     __ beq(j_rarg1, t0, is_float);
 374     __ mv(t0, (u1)T_DOUBLE);
 375     __ beq(j_rarg1, t0, is_double);
 376 
 377     // handle T_INT case
 378     __ sw(x10, Address(j_rarg2));
 379 
 380     __ BIND(exit);
 381 
 382     // pop parameters
 383     __ addi(esp, fp, sp_after_call_off * wordSize);
 384 
 385 #ifdef ASSERT
 386     // verify that threads correspond
 387     {
 388       Label L, S;
 389       __ ld(t0, thread);
 390       __ bne(xthread, t0, S);
 391       __ get_thread(t0);
 392       __ beq(xthread, t0, L);
 393       __ BIND(S);
 394       __ stop("StubRoutines::call_stub: threads must correspond");
 395       __ BIND(L);
 396     }
 397 #endif
 398 
 399     __ pop_cont_fastpath(xthread);
 400 
 401     // restore callee-save registers
 402     __ fld(f27, f27_save);
 403     __ fld(f26, f26_save);
 404     __ fld(f25, f25_save);
 405     __ fld(f24, f24_save);
 406     __ fld(f23, f23_save);
 407     __ fld(f22, f22_save);
 408     __ fld(f21, f21_save);
 409     __ fld(f20, f20_save);
 410     __ fld(f19, f19_save);
 411     __ fld(f18, f18_save);
 412     __ fld(f9,  f9_save);
 413     __ fld(f8,  f8_save);
 414 
 415     __ ld(x27, x27_save);
 416     __ ld(x26, x26_save);
 417     __ ld(x25, x25_save);
 418     __ ld(x24, x24_save);
 419     __ ld(x23, x23_save);
 420     __ ld(x22, x22_save);
 421     __ ld(x21, x21_save);
 422     __ ld(x20, x20_save);
 423     __ ld(x19, x19_save);
 424     __ ld(x18, x18_save);
 425 
 426     __ ld(x9, x9_save);
 427 
 428     // restore frm
 429     Label skip_fsrm;
 430     __ ld(t0, frm_save);
 431     __ frrm(t1);
 432     __ beq(t0, t1, skip_fsrm);
 433     __ fsrm(t0);
 434     __ bind(skip_fsrm);
 435 
 436     __ ld(c_rarg0, call_wrapper);
 437     __ ld(c_rarg1, result);
 438     __ ld(c_rarg2, result_type);
 439     __ ld(c_rarg3, method);
 440     __ ld(c_rarg4, entry_point);
 441     __ ld(c_rarg5, parameters);
 442     __ ld(c_rarg6, parameter_size);
 443     __ ld(c_rarg7, thread);
 444 
 445     // leave frame and return to caller
 446     __ leave();
 447     __ ret();
 448 
 449     // handle return types different from T_INT
 450 
 451     __ BIND(is_long);
 452     __ sd(x10, Address(j_rarg2, 0));
 453     __ j(exit);
 454 
 455     __ BIND(is_float);
 456     __ fsw(j_farg0, Address(j_rarg2, 0), t0);
 457     __ j(exit);
 458 
 459     __ BIND(is_double);
 460     __ fsd(j_farg0, Address(j_rarg2, 0), t0);
 461     __ j(exit);
 462 
 463     return start;
 464   }
 465 
 466   // Return point for a Java call if there's an exception thrown in
 467   // Java code.  The exception is caught and transformed into a
 468   // pending exception stored in JavaThread that can be tested from
 469   // within the VM.
 470   //
 471   // Note: Usually the parameters are removed by the callee. In case
 472   // of an exception crossing an activation frame boundary, that is
 473   // not the case if the callee is compiled code => need to setup the
 474   // sp.
 475   //
 476   // x10: exception oop
 477 
 478   address generate_catch_exception() {
 479     StubId stub_id = StubId::stubgen_catch_exception_id;
 480     StubCodeMark mark(this, stub_id);
 481     address start = __ pc();
 482 
 483     // same as in generate_call_stub():
 484     const Address thread(fp, thread_off * wordSize);
 485 
 486 #ifdef ASSERT
 487     // verify that threads correspond
 488     {
 489       Label L, S;
 490       __ ld(t0, thread);
 491       __ bne(xthread, t0, S);
 492       __ get_thread(t0);
 493       __ beq(xthread, t0, L);
 494       __ bind(S);
 495       __ stop("StubRoutines::catch_exception: threads must correspond");
 496       __ bind(L);
 497     }
 498 #endif
 499 
 500     // set pending exception
 501     __ verify_oop(x10);
 502 
 503     __ sd(x10, Address(xthread, Thread::pending_exception_offset()));
 504     __ mv(t0, (address)__FILE__);
 505     __ sd(t0, Address(xthread, Thread::exception_file_offset()));
 506     __ mv(t0, (int)__LINE__);
 507     __ sw(t0, Address(xthread, Thread::exception_line_offset()));
 508 
 509     // complete return to VM
 510     assert(StubRoutines::_call_stub_return_address != nullptr,
 511            "_call_stub_return_address must have been generated before");
 512     __ j(RuntimeAddress(StubRoutines::_call_stub_return_address));
 513 
 514     return start;
 515   }
 516 
 517   // Continuation point for runtime calls returning with a pending
 518   // exception.  The pending exception check happened in the runtime
 519   // or native call stub.  The pending exception in Thread is
 520   // converted into a Java-level exception.
 521   //
 522   // Contract with Java-level exception handlers:
 523   // x10: exception
 524   // x13: throwing pc
 525   //
 526   // NOTE: At entry of this stub, exception-pc must be in RA !!
 527 
 528   // NOTE: this is always used as a jump target within generated code
 529   // so it just needs to be generated code with no x86 prolog
 530 
 531   address generate_forward_exception() {
 532     StubId stub_id = StubId::stubgen_forward_exception_id;
 533     StubCodeMark mark(this, stub_id);
 534     address start = __ pc();
 535 
 536     // Upon entry, RA points to the return address returning into
 537     // Java (interpreted or compiled) code; i.e., the return address
 538     // becomes the throwing pc.
 539     //
 540     // Arguments pushed before the runtime call are still on the stack
 541     // but the exception handler will reset the stack pointer ->
 542     // ignore them.  A potential result in registers can be ignored as
 543     // well.
 544 
 545 #ifdef ASSERT
 546     // make sure this code is only executed if there is a pending exception
 547     {
 548       Label L;
 549       __ ld(t0, Address(xthread, Thread::pending_exception_offset()));
 550       __ bnez(t0, L);
 551       __ stop("StubRoutines::forward exception: no pending exception (1)");
 552       __ bind(L);
 553     }
 554 #endif
 555 
 556     // compute exception handler into x9
 557 
 558     // call the VM to find the handler address associated with the
 559     // caller address. pass thread in x10 and caller pc (ret address)
 560     // in x11. n.b. the caller pc is in ra, unlike x86 where it is on
 561     // the stack.
 562     __ mv(c_rarg1, ra);
 563     // ra will be trashed by the VM call so we move it to x9
 564     // (callee-saved) because we also need to pass it to the handler
 565     // returned by this call.
 566     __ mv(x9, ra);
 567     BLOCK_COMMENT("call exception_handler_for_return_address");
 568     __ call_VM_leaf(CAST_FROM_FN_PTR(address,
 569                          SharedRuntime::exception_handler_for_return_address),
 570                     xthread, c_rarg1);
 571     // we should not really care that ra is no longer the callee
 572     // address. we saved the value the handler needs in x9 so we can
 573     // just copy it to x13. however, the C2 handler will push its own
 574     // frame and then calls into the VM and the VM code asserts that
 575     // the PC for the frame above the handler belongs to a compiled
 576     // Java method. So, we restore ra here to satisfy that assert.
 577     __ mv(ra, x9);
 578     // setup x10 & x13 & clear pending exception
 579     __ mv(x13, x9);
 580     __ mv(x9, x10);
 581     __ ld(x10, Address(xthread, Thread::pending_exception_offset()));
 582     __ sd(zr, Address(xthread, Thread::pending_exception_offset()));
 583 
 584 #ifdef ASSERT
 585     // make sure exception is set
 586     {
 587       Label L;
 588       __ bnez(x10, L);
 589       __ stop("StubRoutines::forward exception: no pending exception (2)");
 590       __ bind(L);
 591     }
 592 #endif
 593 
 594     // continue at exception handler
 595     // x10: exception
 596     // x13: throwing pc
 597     // x9: exception handler
 598     __ verify_oop(x10);
 599     __ jr(x9);
 600 
 601     return start;
 602   }
 603 
 604   // Non-destructive plausibility checks for oops
 605   //
 606   // Arguments:
 607   //    x10: oop to verify
 608   //    t0: error message
 609   //
 610   // Stack after saving c_rarg3:
 611   //    [tos + 0]: saved c_rarg3
 612   //    [tos + 1]: saved c_rarg2
 613   //    [tos + 2]: saved ra
 614   //    [tos + 3]: saved t1
 615   //    [tos + 4]: saved x10
 616   //    [tos + 5]: saved t0
 617   address generate_verify_oop() {
 618 
 619     StubId stub_id = StubId::stubgen_verify_oop_id;
 620     StubCodeMark mark(this, stub_id);
 621     address start = __ pc();
 622 
 623     Label exit, error;
 624 
 625     __ push_reg(RegSet::of(c_rarg2, c_rarg3), sp); // save c_rarg2 and c_rarg3
 626 
 627     __ la(c_rarg2, ExternalAddress((address) StubRoutines::verify_oop_count_addr()));
 628     __ ld(c_rarg3, Address(c_rarg2));
 629     __ addi(c_rarg3, c_rarg3, 1);
 630     __ sd(c_rarg3, Address(c_rarg2));
 631 
 632     // object is in x10
 633     // make sure object is 'reasonable'
 634     __ beqz(x10, exit); // if obj is null it is OK
 635 
 636     BarrierSetAssembler* bs_asm = BarrierSet::barrier_set()->barrier_set_assembler();
 637     bs_asm->check_oop(_masm, x10, c_rarg2, c_rarg3, error);
 638 
 639     // return if everything seems ok
 640     __ bind(exit);
 641 
 642     __ pop_reg(RegSet::of(c_rarg2, c_rarg3), sp);  // pop c_rarg2 and c_rarg3
 643     __ ret();
 644 
 645     // handle errors
 646     __ bind(error);
 647     __ pop_reg(RegSet::of(c_rarg2, c_rarg3), sp); // pop c_rarg2 and c_rarg3
 648 
 649     __ push_reg(RegSet::range(x0, x31), sp);
 650     // debug(char* msg, int64_t pc, int64_t regs[])
 651     __ mv(c_rarg0, t0);             // pass address of error message
 652     __ mv(c_rarg1, ra);             // pass return address
 653     __ mv(c_rarg2, sp);             // pass address of regs on stack
 654 #ifndef PRODUCT
 655     assert(frame::arg_reg_save_area_bytes == 0, "not expecting frame reg save area");
 656 #endif
 657     BLOCK_COMMENT("call MacroAssembler::debug");
 658     __ rt_call(CAST_FROM_FN_PTR(address, MacroAssembler::debug64));
 659     __ ebreak();
 660 
 661     return start;
 662   }
 663 
 664   // The inner part of zero_words().
 665   //
 666   // Inputs:
 667   // x28: the HeapWord-aligned base address of an array to zero.
 668   // x29: the count in HeapWords, x29 > 0.
 669   //
 670   // Returns x28 and x29, adjusted for the caller to clear.
 671   // x28: the base address of the tail of words left to clear.
 672   // x29: the number of words in the tail.
 673   //      x29 < MacroAssembler::zero_words_block_size.
 674 
 675   address generate_zero_blocks() {
 676     Label done;
 677 
 678     const Register base = x28, cnt = x29, tmp1 = x30, tmp2 = x31;
 679 
 680     __ align(CodeEntryAlignment);
 681     StubId stub_id = StubId::stubgen_zero_blocks_id;
 682     StubCodeMark mark(this, stub_id);
 683     address start = __ pc();
 684 
 685     if (UseBlockZeroing) {
 686       int zicboz_block_size = VM_Version::zicboz_block_size.value();
 687       // Ensure count >= 2 * zicboz_block_size so that it still deserves
 688       // a cbo.zero after alignment.
 689       Label small;
 690       int low_limit = MAX2(2 * zicboz_block_size, (int)BlockZeroingLowLimit) / wordSize;
 691       __ mv(tmp1, low_limit);
 692       __ blt(cnt, tmp1, small);
 693       __ zero_dcache_blocks(base, cnt, tmp1, tmp2);
 694       __ bind(small);
 695     }
 696 
 697     {
 698       // Clear the remaining blocks.
 699       Label loop;
 700       __ mv(tmp1, MacroAssembler::zero_words_block_size);
 701       __ blt(cnt, tmp1, done);
 702       __ bind(loop);
 703       for (int i = 0; i < MacroAssembler::zero_words_block_size; i++) {
 704         __ sd(zr, Address(base, i * wordSize));
 705       }
 706       __ addi(base, base, MacroAssembler::zero_words_block_size * wordSize);
 707       __ subi(cnt, cnt, MacroAssembler::zero_words_block_size);
 708       __ bge(cnt, tmp1, loop);
 709       __ bind(done);
 710     }
 711 
 712     __ ret();
 713 
 714     return start;
 715   }
 716 
 717   typedef enum {
 718     copy_forwards = 1,
 719     copy_backwards = -1
 720   } copy_direction;
 721 
 722   // Bulk copy of blocks of 8 words.
 723   //
 724   // count is a count of words.
 725   //
 726   // Precondition: count >= 8
 727   //
 728   // Postconditions:
 729   //
 730   // The least significant bit of count contains the remaining count
 731   // of words to copy.  The rest of count is trash.
 732   //
 733   // s and d are adjusted to point to the remaining words to copy
 734   //
 735   address generate_copy_longs(StubId stub_id, Register s, Register d, Register count) {
 736     BasicType type;
 737     copy_direction direction;
 738     switch (stub_id) {
 739     case StubId::stubgen_copy_byte_f_id:
 740       direction = copy_forwards;
 741       type = T_BYTE;
 742       break;
 743     case StubId::stubgen_copy_byte_b_id:
 744       direction = copy_backwards;
 745       type = T_BYTE;
 746       break;
 747     default:
 748       ShouldNotReachHere();
 749     }
 750     int unit = wordSize * direction;
 751     int bias = wordSize;
 752 
 753     const Register tmp_reg0 = x13, tmp_reg1 = x14, tmp_reg2 = x15, tmp_reg3 = x16,
 754       tmp_reg4 = x17, tmp_reg5 = x7, tmp_reg6 = x28, tmp_reg7 = x29;
 755 
 756     const Register stride = x30;
 757 
 758     assert_different_registers(t0, tmp_reg0, tmp_reg1, tmp_reg2, tmp_reg3,
 759       tmp_reg4, tmp_reg5, tmp_reg6, tmp_reg7);
 760     assert_different_registers(s, d, count, t0);
 761 
 762     Label again, drain;
 763     StubCodeMark mark(this, stub_id);
 764     __ align(CodeEntryAlignment);
 765     address start = __ pc();
 766 
 767     if (direction == copy_forwards) {
 768       __ sub(s, s, bias);
 769       __ sub(d, d, bias);
 770     }
 771 
 772 #ifdef ASSERT
 773     // Make sure we are never given < 8 words
 774     {
 775       Label L;
 776 
 777       __ mv(t0, 8);
 778       __ bge(count, t0, L);
 779       __ stop("genrate_copy_longs called with < 8 words");
 780       __ bind(L);
 781     }
 782 #endif
 783 
 784     __ ld(tmp_reg0, Address(s, 1 * unit));
 785     __ ld(tmp_reg1, Address(s, 2 * unit));
 786     __ ld(tmp_reg2, Address(s, 3 * unit));
 787     __ ld(tmp_reg3, Address(s, 4 * unit));
 788     __ ld(tmp_reg4, Address(s, 5 * unit));
 789     __ ld(tmp_reg5, Address(s, 6 * unit));
 790     __ ld(tmp_reg6, Address(s, 7 * unit));
 791     __ ld(tmp_reg7, Address(s, 8 * unit));
 792     __ addi(s, s, 8 * unit);
 793 
 794     __ subi(count, count, 16);
 795     __ bltz(count, drain);
 796 
 797     __ bind(again);
 798 
 799     __ sd(tmp_reg0, Address(d, 1 * unit));
 800     __ sd(tmp_reg1, Address(d, 2 * unit));
 801     __ sd(tmp_reg2, Address(d, 3 * unit));
 802     __ sd(tmp_reg3, Address(d, 4 * unit));
 803     __ sd(tmp_reg4, Address(d, 5 * unit));
 804     __ sd(tmp_reg5, Address(d, 6 * unit));
 805     __ sd(tmp_reg6, Address(d, 7 * unit));
 806     __ sd(tmp_reg7, Address(d, 8 * unit));
 807 
 808     __ ld(tmp_reg0, Address(s, 1 * unit));
 809     __ ld(tmp_reg1, Address(s, 2 * unit));
 810     __ ld(tmp_reg2, Address(s, 3 * unit));
 811     __ ld(tmp_reg3, Address(s, 4 * unit));
 812     __ ld(tmp_reg4, Address(s, 5 * unit));
 813     __ ld(tmp_reg5, Address(s, 6 * unit));
 814     __ ld(tmp_reg6, Address(s, 7 * unit));
 815     __ ld(tmp_reg7, Address(s, 8 * unit));
 816 
 817     __ addi(s, s, 8 * unit);
 818     __ addi(d, d, 8 * unit);
 819 
 820     __ subi(count, count, 8);
 821     __ bgez(count, again);
 822 
 823     // Drain
 824     __ bind(drain);
 825 
 826     __ sd(tmp_reg0, Address(d, 1 * unit));
 827     __ sd(tmp_reg1, Address(d, 2 * unit));
 828     __ sd(tmp_reg2, Address(d, 3 * unit));
 829     __ sd(tmp_reg3, Address(d, 4 * unit));
 830     __ sd(tmp_reg4, Address(d, 5 * unit));
 831     __ sd(tmp_reg5, Address(d, 6 * unit));
 832     __ sd(tmp_reg6, Address(d, 7 * unit));
 833     __ sd(tmp_reg7, Address(d, 8 * unit));
 834     __ addi(d, d, 8 * unit);
 835 
 836     {
 837       Label L1, L2;
 838       __ test_bit(t0, count, 2);
 839       __ beqz(t0, L1);
 840 
 841       __ ld(tmp_reg0, Address(s, 1 * unit));
 842       __ ld(tmp_reg1, Address(s, 2 * unit));
 843       __ ld(tmp_reg2, Address(s, 3 * unit));
 844       __ ld(tmp_reg3, Address(s, 4 * unit));
 845       __ addi(s, s, 4 * unit);
 846 
 847       __ sd(tmp_reg0, Address(d, 1 * unit));
 848       __ sd(tmp_reg1, Address(d, 2 * unit));
 849       __ sd(tmp_reg2, Address(d, 3 * unit));
 850       __ sd(tmp_reg3, Address(d, 4 * unit));
 851       __ addi(d, d, 4 * unit);
 852 
 853       __ bind(L1);
 854 
 855       if (direction == copy_forwards) {
 856         __ addi(s, s, bias);
 857         __ addi(d, d, bias);
 858       }
 859 
 860       __ test_bit(t0, count, 1);
 861       __ beqz(t0, L2);
 862       if (direction == copy_backwards) {
 863         __ addi(s, s, 2 * unit);
 864         __ ld(tmp_reg0, Address(s));
 865         __ ld(tmp_reg1, Address(s, wordSize));
 866         __ addi(d, d, 2 * unit);
 867         __ sd(tmp_reg0, Address(d));
 868         __ sd(tmp_reg1, Address(d, wordSize));
 869       } else {
 870         __ ld(tmp_reg0, Address(s));
 871         __ ld(tmp_reg1, Address(s, wordSize));
 872         __ addi(s, s, 2 * unit);
 873         __ sd(tmp_reg0, Address(d));
 874         __ sd(tmp_reg1, Address(d, wordSize));
 875         __ addi(d, d, 2 * unit);
 876       }
 877       __ bind(L2);
 878     }
 879 
 880     __ ret();
 881 
 882     return start;
 883   }
 884 
 885   typedef void (MacroAssembler::*copy_insn)(Register Rd, const Address &adr, Register temp);
 886 
 887   void copy_memory_v(Register s, Register d, Register count, int step) {
 888     bool is_backward = step < 0;
 889     int granularity = g_uabs(step);
 890 
 891     const Register src = x30, dst = x31, vl = x14, cnt = x15, tmp1 = x16, tmp2 = x17;
 892     assert_different_registers(s, d, cnt, vl, tmp1, tmp2);
 893     Assembler::SEW sew = Assembler::elembytes_to_sew(granularity);
 894     Label loop_forward, loop_backward, done;
 895 
 896     __ mv(dst, d);
 897     __ mv(src, s);
 898     __ mv(cnt, count);
 899 
 900     __ bind(loop_forward);
 901     __ vsetvli(vl, cnt, sew, Assembler::m8);
 902     if (is_backward) {
 903       __ bne(vl, cnt, loop_backward);
 904     }
 905 
 906     __ vlex_v(v0, src, sew);
 907     __ sub(cnt, cnt, vl);
 908     if (sew != Assembler::e8) {
 909       // when sew == e8 (e.g., elem size is 1 byte), slli R, R, 0 is a nop and unnecessary
 910       __ slli(vl, vl, sew);
 911     }
 912     __ add(src, src, vl);
 913 
 914     __ vsex_v(v0, dst, sew);
 915     __ add(dst, dst, vl);
 916     __ bnez(cnt, loop_forward);
 917 
 918     if (is_backward) {
 919       __ j(done);
 920 
 921       __ bind(loop_backward);
 922       __ sub(t0, cnt, vl);
 923       if (sew != Assembler::e8) {
 924         // when sew == e8 (e.g., elem size is 1 byte), slli R, R, 0 is a nop and unnecessary
 925         __ slli(t0, t0, sew);
 926       }
 927       __ add(tmp1, s, t0);
 928       __ vlex_v(v0, tmp1, sew);
 929       __ add(tmp2, d, t0);
 930       __ vsex_v(v0, tmp2, sew);
 931       __ sub(cnt, cnt, vl);
 932       __ bnez(cnt, loop_forward);
 933       __ bind(done);
 934     }
 935   }
 936 
 937   // All-singing all-dancing memory copy.
 938   //
 939   // Copy count units of memory from s to d.  The size of a unit is
 940   // step, which can be positive or negative depending on the direction
 941   // of copy.
 942   //
 943   void copy_memory(DecoratorSet decorators, BasicType type, bool is_aligned,
 944                    Register s, Register d, Register count, int step) {
 945     BarrierSetAssembler* bs_asm = BarrierSet::barrier_set()->barrier_set_assembler();
 946     if (UseRVV && (!is_reference_type(type) || bs_asm->supports_rvv_arraycopy())) {
 947       return copy_memory_v(s, d, count, step);
 948     }
 949 
 950     bool is_backwards = step < 0;
 951     int granularity = g_uabs(step);
 952 
 953     const Register src = x30, dst = x31, cnt = x15, tmp3 = x16, tmp4 = x17, tmp5 = x14, tmp6 = x13;
 954     const Register gct1 = x28, gct2 = x29, gct3 = t2;
 955 
 956     Label same_aligned;
 957     Label copy_big, copy32_loop, copy8_loop, copy_small, done;
 958 
 959     // The size of copy32_loop body increases significantly with ZGC GC barriers.
 960     // Need conditional far branches to reach a point beyond the loop in this case.
 961     bool is_far = UseZGC;
 962 
 963     __ beqz(count, done, is_far);
 964     __ slli(cnt, count, exact_log2(granularity));
 965     if (is_backwards) {
 966       __ add(src, s, cnt);
 967       __ add(dst, d, cnt);
 968     } else {
 969       __ mv(src, s);
 970       __ mv(dst, d);
 971     }
 972 
 973     if (is_aligned) {
 974       __ subi(t0, cnt, 32);
 975       __ bgez(t0, copy32_loop);
 976       __ subi(t0, cnt, 8);
 977       __ bgez(t0, copy8_loop, is_far);
 978       __ j(copy_small);
 979     } else {
 980       __ mv(t0, 16);
 981       __ blt(cnt, t0, copy_small, is_far);
 982 
 983       __ xorr(t0, src, dst);
 984       __ andi(t0, t0, 0b111);
 985       __ bnez(t0, copy_small, is_far);
 986 
 987       __ bind(same_aligned);
 988       __ andi(t0, src, 0b111);
 989       __ beqz(t0, copy_big);
 990       if (is_backwards) {
 991         __ addi(src, src, step);
 992         __ addi(dst, dst, step);
 993       }
 994       bs_asm->copy_load_at(_masm, decorators, type, granularity, tmp3, Address(src), gct1);
 995       bs_asm->copy_store_at(_masm, decorators, type, granularity, Address(dst), tmp3, gct1, gct2, gct3);
 996       if (!is_backwards) {
 997         __ addi(src, src, step);
 998         __ addi(dst, dst, step);
 999       }
1000       __ subi(cnt, cnt, granularity);
1001       __ beqz(cnt, done, is_far);
1002       __ j(same_aligned);
1003 
1004       __ bind(copy_big);
1005       __ mv(t0, 32);
1006       __ blt(cnt, t0, copy8_loop, is_far);
1007     }
1008 
1009     __ bind(copy32_loop);
1010     if (is_backwards) {
1011       __ subi(src, src, wordSize * 4);
1012       __ subi(dst, dst, wordSize * 4);
1013     }
1014     // we first load 32 bytes, then write it, so the direction here doesn't matter
1015     bs_asm->copy_load_at(_masm, decorators, type, 8, tmp3, Address(src),     gct1);
1016     bs_asm->copy_load_at(_masm, decorators, type, 8, tmp4, Address(src, 8),  gct1);
1017     bs_asm->copy_load_at(_masm, decorators, type, 8, tmp5, Address(src, 16), gct1);
1018     bs_asm->copy_load_at(_masm, decorators, type, 8, tmp6, Address(src, 24), gct1);
1019 
1020     bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst),     tmp3, gct1, gct2, gct3);
1021     bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst, 8),  tmp4, gct1, gct2, gct3);
1022     bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst, 16), tmp5, gct1, gct2, gct3);
1023     bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst, 24), tmp6, gct1, gct2, gct3);
1024 
1025     if (!is_backwards) {
1026       __ addi(src, src, wordSize * 4);
1027       __ addi(dst, dst, wordSize * 4);
1028     }
1029     __ subi(t0, cnt, 32 + wordSize * 4);
1030     __ subi(cnt, cnt, wordSize * 4);
1031     __ bgez(t0, copy32_loop); // cnt >= 32, do next loop
1032 
1033     __ beqz(cnt, done); // if that's all - done
1034 
1035     __ subi(t0, cnt, 8); // if not - copy the reminder
1036     __ bltz(t0, copy_small); // cnt < 8, go to copy_small, else fall through to copy8_loop
1037 
1038     __ bind(copy8_loop);
1039     if (is_backwards) {
1040       __ subi(src, src, wordSize);
1041       __ subi(dst, dst, wordSize);
1042     }
1043     bs_asm->copy_load_at(_masm, decorators, type, 8, tmp3, Address(src), gct1);
1044     bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst), tmp3, gct1, gct2, gct3);
1045 
1046     if (!is_backwards) {
1047       __ addi(src, src, wordSize);
1048       __ addi(dst, dst, wordSize);
1049     }
1050     __ subi(t0, cnt, 8 + wordSize);
1051     __ subi(cnt, cnt, wordSize);
1052     __ bgez(t0, copy8_loop); // cnt >= 8, do next loop
1053 
1054     __ beqz(cnt, done); // if that's all - done
1055 
1056     __ bind(copy_small);
1057     if (is_backwards) {
1058       __ addi(src, src, step);
1059       __ addi(dst, dst, step);
1060     }
1061 
1062     bs_asm->copy_load_at(_masm, decorators, type, granularity, tmp3, Address(src), gct1);
1063     bs_asm->copy_store_at(_masm, decorators, type, granularity, Address(dst), tmp3, gct1, gct2, gct3);
1064 
1065     if (!is_backwards) {
1066       __ addi(src, src, step);
1067       __ addi(dst, dst, step);
1068     }
1069     __ subi(cnt, cnt, granularity);
1070     __ bgtz(cnt, copy_small);
1071 
1072     __ bind(done);
1073   }
1074 
1075   // Scan over array at a for count oops, verifying each one.
1076   // Preserves a and count, clobbers t0 and t1.
1077   void verify_oop_array(size_t size, Register a, Register count, Register temp) {
1078     Label loop, end;
1079     __ mv(t1, zr);
1080     __ slli(t0, count, exact_log2(size));
1081     __ bind(loop);
1082     __ bgeu(t1, t0, end);
1083 
1084     __ add(temp, a, t1);
1085     if (size == (size_t)wordSize) {
1086       __ ld(temp, Address(temp, 0));
1087       __ verify_oop(temp);
1088     } else {
1089       __ lwu(temp, Address(temp, 0));
1090       __ decode_heap_oop(temp); // calls verify_oop
1091     }
1092     __ add(t1, t1, size);
1093     __ j(loop);
1094     __ bind(end);
1095   }
1096 
1097   // Arguments:
1098   //   stub_id - is used to name the stub and identify all details of
1099   //             how to perform the copy.
1100   //
1101   //   nopush_entry - is assigned to the stub's post push entry point
1102   //                  unless it is null
1103   //
1104   // Inputs:
1105   //   c_rarg0   - source array address
1106   //   c_rarg1   - destination array address
1107   //   c_rarg2   - element count, treated as ssize_t, can be zero
1108   //
1109   // If 'from' and/or 'to' are aligned on 4-byte boundaries, we let
1110   // the hardware handle it.  The two dwords within qwords that span
1111   // cache line boundaries will still be loaded and stored atomically.
1112   //
1113   // Side Effects: nopush_entry is set to the (post push) entry point
1114   //               so it can be used by the corresponding conjoint
1115   //               copy method
1116   //
1117   address generate_disjoint_copy(StubId stub_id, address* nopush_entry) {
1118     size_t size;
1119     bool aligned;
1120     bool is_oop;
1121     bool dest_uninitialized;
1122     switch (stub_id) {
1123     case StubId::stubgen_jbyte_disjoint_arraycopy_id:
1124       size = sizeof(jbyte);
1125       aligned = false;
1126       is_oop = false;
1127       dest_uninitialized = false;
1128       break;
1129     case StubId::stubgen_arrayof_jbyte_disjoint_arraycopy_id:
1130       size = sizeof(jbyte);
1131       aligned = true;
1132       is_oop = false;
1133       dest_uninitialized = false;
1134       break;
1135     case StubId::stubgen_jshort_disjoint_arraycopy_id:
1136       size = sizeof(jshort);
1137       aligned = false;
1138       is_oop = false;
1139       dest_uninitialized = false;
1140       break;
1141     case StubId::stubgen_arrayof_jshort_disjoint_arraycopy_id:
1142       size = sizeof(jshort);
1143       aligned = true;
1144       is_oop = false;
1145       dest_uninitialized = false;
1146       break;
1147     case StubId::stubgen_jint_disjoint_arraycopy_id:
1148       size = sizeof(jint);
1149       aligned = false;
1150       is_oop = false;
1151       dest_uninitialized = false;
1152       break;
1153     case StubId::stubgen_arrayof_jint_disjoint_arraycopy_id:
1154       size = sizeof(jint);
1155       aligned = true;
1156       is_oop = false;
1157       dest_uninitialized = false;
1158       break;
1159     case StubId::stubgen_jlong_disjoint_arraycopy_id:
1160       // since this is always aligned we can (should!) use the same
1161       // stub as for case arrayof_jlong_disjoint_arraycopy
1162       ShouldNotReachHere();
1163       break;
1164     case StubId::stubgen_arrayof_jlong_disjoint_arraycopy_id:
1165       size = sizeof(jlong);
1166       aligned = true;
1167       is_oop = false;
1168       dest_uninitialized = false;
1169       break;
1170     case StubId::stubgen_oop_disjoint_arraycopy_id:
1171       size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1172       aligned = !UseCompressedOops;
1173       is_oop = true;
1174       dest_uninitialized = false;
1175       break;
1176     case StubId::stubgen_arrayof_oop_disjoint_arraycopy_id:
1177       size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1178       aligned = !UseCompressedOops;
1179       is_oop = true;
1180       dest_uninitialized = false;
1181       break;
1182     case StubId::stubgen_oop_disjoint_arraycopy_uninit_id:
1183       size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1184       aligned = !UseCompressedOops;
1185       is_oop = true;
1186       dest_uninitialized = true;
1187       break;
1188     case StubId::stubgen_arrayof_oop_disjoint_arraycopy_uninit_id:
1189       size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1190       aligned = !UseCompressedOops;
1191       is_oop = true;
1192       dest_uninitialized = true;
1193       break;
1194     default:
1195       ShouldNotReachHere();
1196       break;
1197     }
1198 
1199     const Register s = c_rarg0, d = c_rarg1, count = c_rarg2;
1200     RegSet saved_reg = RegSet::of(s, d, count);
1201     __ align(CodeEntryAlignment);
1202     StubCodeMark mark(this, stub_id);
1203     address start = __ pc();
1204     __ enter();
1205 
1206     if (nopush_entry != nullptr) {
1207      *nopush_entry = __ pc();
1208       // caller can pass a 64-bit byte count here (from Unsafe.copyMemory)
1209       BLOCK_COMMENT("Entry:");
1210     }
1211 
1212     DecoratorSet decorators = IN_HEAP | IS_ARRAY | ARRAYCOPY_DISJOINT;
1213     if (dest_uninitialized) {
1214       decorators |= IS_DEST_UNINITIALIZED;
1215     }
1216     if (aligned) {
1217       decorators |= ARRAYCOPY_ALIGNED;
1218     }
1219 
1220     BarrierSetAssembler *bs = BarrierSet::barrier_set()->barrier_set_assembler();
1221     bs->arraycopy_prologue(_masm, decorators, is_oop, s, d, count, saved_reg);
1222 
1223     if (is_oop) {
1224       // save regs before copy_memory
1225       __ push_reg(RegSet::of(d, count), sp);
1226     }
1227 
1228     {
1229       // UnsafeMemoryAccess page error: continue after unsafe access
1230       bool add_entry = !is_oop && (!aligned || sizeof(jlong) == size);
1231       UnsafeMemoryAccessMark umam(this, add_entry, true);
1232       copy_memory(decorators, is_oop ? T_OBJECT : T_BYTE, aligned, s, d, count, size);
1233     }
1234 
1235     if (is_oop) {
1236       __ pop_reg(RegSet::of(d, count), sp);
1237       if (VerifyOops) {
1238         verify_oop_array(size, d, count, t2);
1239       }
1240     }
1241 
1242     bs->arraycopy_epilogue(_masm, decorators, is_oop, d, count, t0);
1243 
1244     __ leave();
1245     __ mv(x10, zr); // return 0
1246     __ ret();
1247     return start;
1248   }
1249 
1250   // Arguments:
1251   //   stub_id - is used to name the stub and identify all details of
1252   //             how to perform the copy.
1253   //
1254   //   nooverlap_target - identifes the (post push) entry for the
1255   //             corresponding disjoint copy routine which can be
1256   //             jumped to if the ranges do not actually overlap
1257   //
1258   //   nopush_entry - is assigned to the stub's post push entry point
1259   //                 unless it is null
1260   //
1261   // Inputs:
1262   //   c_rarg0   - source array address
1263   //   c_rarg1   - destination array address
1264   //   c_rarg2   - element count, treated as ssize_t, can be zero
1265   //
1266   // If 'from' and/or 'to' are aligned on 4-byte boundaries, we let
1267   // the hardware handle it.  The two dwords within qwords that span
1268   // cache line boundaries will still be loaded and stored atomically.
1269   //
1270   // Side Effects:
1271   //   nopush_entry is set to the no-overlap entry point so it can be
1272   //   used by some other conjoint copy method
1273   //
1274   address generate_conjoint_copy(StubId stub_id, address nooverlap_target, address *nopush_entry) {
1275     const Register s = c_rarg0, d = c_rarg1, count = c_rarg2;
1276     RegSet saved_regs = RegSet::of(s, d, count);
1277     int size;
1278     bool aligned;
1279     bool is_oop;
1280     bool dest_uninitialized;
1281     switch (stub_id) {
1282     case StubId::stubgen_jbyte_arraycopy_id:
1283       size = sizeof(jbyte);
1284       aligned = false;
1285       is_oop = false;
1286       dest_uninitialized = false;
1287       break;
1288     case StubId::stubgen_arrayof_jbyte_arraycopy_id:
1289       size = sizeof(jbyte);
1290       aligned = true;
1291       is_oop = false;
1292       dest_uninitialized = false;
1293       break;
1294     case StubId::stubgen_jshort_arraycopy_id:
1295       size = sizeof(jshort);
1296       aligned = false;
1297       is_oop = false;
1298       dest_uninitialized = false;
1299       break;
1300     case StubId::stubgen_arrayof_jshort_arraycopy_id:
1301       size = sizeof(jshort);
1302       aligned = true;
1303       is_oop = false;
1304       dest_uninitialized = false;
1305       break;
1306     case StubId::stubgen_jint_arraycopy_id:
1307       size = sizeof(jint);
1308       aligned = false;
1309       is_oop = false;
1310       dest_uninitialized = false;
1311       break;
1312     case StubId::stubgen_arrayof_jint_arraycopy_id:
1313       size = sizeof(jint);
1314       aligned = true;
1315       is_oop = false;
1316       dest_uninitialized = false;
1317       break;
1318     case StubId::stubgen_jlong_arraycopy_id:
1319       // since this is always aligned we can (should!) use the same
1320       // stub as for case arrayof_jlong_disjoint_arraycopy
1321       ShouldNotReachHere();
1322       break;
1323     case StubId::stubgen_arrayof_jlong_arraycopy_id:
1324       size = sizeof(jlong);
1325       aligned = true;
1326       is_oop = false;
1327       dest_uninitialized = false;
1328       break;
1329     case StubId::stubgen_oop_arraycopy_id:
1330       size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1331       aligned = !UseCompressedOops;
1332       is_oop = true;
1333       dest_uninitialized = false;
1334       break;
1335     case StubId::stubgen_arrayof_oop_arraycopy_id:
1336       size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1337       aligned = !UseCompressedOops;
1338       is_oop = true;
1339       dest_uninitialized = false;
1340       break;
1341     case StubId::stubgen_oop_arraycopy_uninit_id:
1342       size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1343       aligned = !UseCompressedOops;
1344       is_oop = true;
1345       dest_uninitialized = true;
1346       break;
1347     case StubId::stubgen_arrayof_oop_arraycopy_uninit_id:
1348       size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1349       aligned = !UseCompressedOops;
1350       is_oop = true;
1351       dest_uninitialized = true;
1352       break;
1353     default:
1354       ShouldNotReachHere();
1355     }
1356 
1357     StubCodeMark mark(this, stub_id);
1358     address start = __ pc();
1359     __ enter();
1360 
1361     if (nopush_entry != nullptr) {
1362       *nopush_entry = __ pc();
1363       // caller can pass a 64-bit byte count here (from Unsafe.copyMemory)
1364       BLOCK_COMMENT("Entry:");
1365     }
1366 
1367     // use fwd copy when (d-s) above_equal (count*size)
1368     __ sub(t0, d, s);
1369     __ slli(t1, count, exact_log2(size));
1370     Label L_continue;
1371     __ bltu(t0, t1, L_continue);
1372     __ j(RuntimeAddress(nooverlap_target));
1373     __ bind(L_continue);
1374 
1375     DecoratorSet decorators = IN_HEAP | IS_ARRAY;
1376     if (dest_uninitialized) {
1377       decorators |= IS_DEST_UNINITIALIZED;
1378     }
1379     if (aligned) {
1380       decorators |= ARRAYCOPY_ALIGNED;
1381     }
1382 
1383     BarrierSetAssembler *bs = BarrierSet::barrier_set()->barrier_set_assembler();
1384     bs->arraycopy_prologue(_masm, decorators, is_oop, s, d, count, saved_regs);
1385 
1386     if (is_oop) {
1387       // save regs before copy_memory
1388       __ push_reg(RegSet::of(d, count), sp);
1389     }
1390 
1391     {
1392       // UnsafeMemoryAccess page error: continue after unsafe access
1393       bool add_entry = !is_oop && (!aligned || sizeof(jlong) == size);
1394       UnsafeMemoryAccessMark umam(this, add_entry, true);
1395       copy_memory(decorators, is_oop ? T_OBJECT : T_BYTE, aligned, s, d, count, -size);
1396     }
1397 
1398     if (is_oop) {
1399       __ pop_reg(RegSet::of(d, count), sp);
1400       if (VerifyOops) {
1401         verify_oop_array(size, d, count, t2);
1402       }
1403     }
1404     bs->arraycopy_epilogue(_masm, decorators, is_oop, d, count, t0);
1405     __ leave();
1406     __ mv(x10, zr); // return 0
1407     __ ret();
1408     return start;
1409   }
1410 
1411   // Helper for generating a dynamic type check.
1412   // Smashes t0, t1.
1413   void generate_type_check(Register sub_klass,
1414                            Register super_check_offset,
1415                            Register super_klass,
1416                            Register result,
1417                            Register tmp1,
1418                            Register tmp2,
1419                            Label& L_success) {
1420     assert_different_registers(sub_klass, super_check_offset, super_klass);
1421 
1422     BLOCK_COMMENT("type_check:");
1423 
1424     Label L_miss;
1425 
1426     __ check_klass_subtype_fast_path(sub_klass, super_klass, noreg, &L_success, &L_miss, nullptr, super_check_offset);
1427     __ check_klass_subtype_slow_path(sub_klass, super_klass, tmp1, tmp2, &L_success, nullptr);
1428 
1429     // Fall through on failure!
1430     __ BIND(L_miss);
1431   }
1432 
1433   //
1434   //  Generate checkcasting array copy stub
1435   //
1436   //  Input:
1437   //    c_rarg0   - source array address
1438   //    c_rarg1   - destination array address
1439   //    c_rarg2   - element count, treated as ssize_t, can be zero
1440   //    c_rarg3   - size_t ckoff (super_check_offset)
1441   //    c_rarg4   - oop ckval (super_klass)
1442   //
1443   //  Output:
1444   //    x10 ==  0  -  success
1445   //    x10 == -1^K - failure, where K is partial transfer count
1446   //
1447   address generate_checkcast_copy(StubId stub_id, address* nopush_entry) {
1448     bool dest_uninitialized;
1449     switch (stub_id) {
1450     case StubId::stubgen_checkcast_arraycopy_id:
1451       dest_uninitialized = false;
1452       break;
1453     case StubId::stubgen_checkcast_arraycopy_uninit_id:
1454       dest_uninitialized = true;
1455       break;
1456     default:
1457       ShouldNotReachHere();
1458     }
1459 
1460     Label L_load_element, L_store_element, L_do_card_marks, L_done, L_done_pop;
1461 
1462     // Input registers (after setup_arg_regs)
1463     const Register from        = c_rarg0;   // source array address
1464     const Register to          = c_rarg1;   // destination array address
1465     const Register count       = c_rarg2;   // elementscount
1466     const Register ckoff       = c_rarg3;   // super_check_offset
1467     const Register ckval       = c_rarg4;   // super_klass
1468 
1469     RegSet wb_pre_saved_regs   = RegSet::range(c_rarg0, c_rarg4);
1470 
1471     // Registers used as temps (x7, x9, x18 are save-on-entry)
1472     const Register count_save  = x19;       // orig elementscount
1473     const Register start_to    = x18;       // destination array start address
1474     const Register copied_oop  = x7;        // actual oop copied
1475     const Register r9_klass    = x9;        // oop._klass
1476 
1477     // Registers used as gc temps (x15, x16, x17 are save-on-call)
1478     const Register gct1 = x15, gct2 = x16, gct3 = x17;
1479 
1480     //---------------------------------------------------------------
1481     // Assembler stub will be used for this call to arraycopy
1482     // if the two arrays are subtypes of Object[] but the
1483     // destination array type is not equal to or a supertype
1484     // of the source type.  Each element must be separately
1485     // checked.
1486 
1487     assert_different_registers(from, to, count, ckoff, ckval, start_to,
1488                                copied_oop, r9_klass, count_save);
1489 
1490     __ align(CodeEntryAlignment);
1491     StubCodeMark mark(this, stub_id);
1492     address start = __ pc();
1493 
1494     __ enter(); // required for proper stackwalking of RuntimeStub frame
1495 
1496     // Caller of this entry point must set up the argument registers.
1497     if (nopush_entry != nullptr) {
1498       *nopush_entry = __ pc();
1499       BLOCK_COMMENT("Entry:");
1500     }
1501 
1502     // Empty array:  Nothing to do
1503     __ beqz(count, L_done);
1504 
1505     __ push_reg(RegSet::of(x7, x9, x18, x19), sp);
1506 
1507 #ifdef ASSERT
1508     BLOCK_COMMENT("assert consistent ckoff/ckval");
1509     // The ckoff and ckval must be mutually consistent,
1510     // even though caller generates both.
1511     { Label L;
1512       int sco_offset = in_bytes(Klass::super_check_offset_offset());
1513       __ lwu(start_to, Address(ckval, sco_offset));
1514       __ beq(ckoff, start_to, L);
1515       __ stop("super_check_offset inconsistent");
1516       __ bind(L);
1517     }
1518 #endif //ASSERT
1519 
1520     DecoratorSet decorators = IN_HEAP | IS_ARRAY | ARRAYCOPY_CHECKCAST | ARRAYCOPY_DISJOINT;
1521     if (dest_uninitialized) {
1522       decorators |= IS_DEST_UNINITIALIZED;
1523     }
1524 
1525     bool is_oop = true;
1526     int element_size = UseCompressedOops ? 4 : 8;
1527 
1528     BarrierSetAssembler *bs = BarrierSet::barrier_set()->barrier_set_assembler();
1529     bs->arraycopy_prologue(_masm, decorators, is_oop, from, to, count, wb_pre_saved_regs);
1530 
1531     // save the original count
1532     __ mv(count_save, count);
1533 
1534     // Copy from low to high addresses
1535     __ mv(start_to, to);              // Save destination array start address
1536     __ j(L_load_element);
1537 
1538     // ======== begin loop ========
1539     // (Loop is rotated; its entry is L_load_element.)
1540     // Loop control:
1541     //   for count to 0 do
1542     //     copied_oop = load_heap_oop(from++)
1543     //     ... generate_type_check ...
1544     //     store_heap_oop(to++, copied_oop)
1545     //   end
1546 
1547     __ align(OptoLoopAlignment);
1548 
1549     __ BIND(L_store_element);
1550     bs->copy_store_at(_masm, decorators, T_OBJECT, element_size,
1551                       Address(to, 0), copied_oop,
1552                       gct1, gct2, gct3);
1553     __ addi(to, to, UseCompressedOops ? 4 : 8);
1554     __ subi(count, count, 1);
1555     __ beqz(count, L_do_card_marks);
1556 
1557     // ======== loop entry is here ========
1558     __ BIND(L_load_element);
1559     bs->copy_load_at(_masm, decorators, T_OBJECT, element_size,
1560                      copied_oop, Address(from, 0),
1561                      gct1);
1562     __ addi(from, from, UseCompressedOops ? 4 : 8);
1563     __ beqz(copied_oop, L_store_element);
1564 
1565     __ load_klass(r9_klass, copied_oop);// query the object klass
1566 
1567     BLOCK_COMMENT("type_check:");
1568     generate_type_check(r9_klass, /*sub_klass*/
1569                         ckoff,    /*super_check_offset*/
1570                         ckval,    /*super_klass*/
1571                         x10,      /*result*/
1572                         gct1,     /*tmp1*/
1573                         gct2,     /*tmp2*/
1574                         L_store_element);
1575 
1576     // Fall through on failure!
1577 
1578     // ======== end loop ========
1579 
1580     // It was a real error; we must depend on the caller to finish the job.
1581     // Register count = remaining oops, count_orig = total oops.
1582     // Emit GC store barriers for the oops we have copied and report
1583     // their number to the caller.
1584 
1585     __ sub(count, count_save, count);     // K = partially copied oop count
1586     __ xori(count, count, -1);            // report (-1^K) to caller
1587     __ beqz(count, L_done_pop);
1588 
1589     __ BIND(L_do_card_marks);
1590     bs->arraycopy_epilogue(_masm, decorators, is_oop, start_to, count_save, t0);
1591 
1592     __ bind(L_done_pop);
1593     __ pop_reg(RegSet::of(x7, x9, x18, x19), sp);
1594     inc_counter_np(SharedRuntime::_checkcast_array_copy_ctr);
1595 
1596     __ bind(L_done);
1597     __ mv(x10, count);
1598     __ leave();
1599     __ ret();
1600 
1601     return start;
1602   }
1603 
1604   // Perform range checks on the proposed arraycopy.
1605   // Kills temp, but nothing else.
1606   // Also, clean the sign bits of src_pos and dst_pos.
1607   void arraycopy_range_checks(Register src,     // source array oop (c_rarg0)
1608                               Register src_pos, // source position (c_rarg1)
1609                               Register dst,     // destination array oo (c_rarg2)
1610                               Register dst_pos, // destination position (c_rarg3)
1611                               Register length,
1612                               Register temp,
1613                               Label& L_failed) {
1614     BLOCK_COMMENT("arraycopy_range_checks:");
1615 
1616     assert_different_registers(t0, temp);
1617 
1618     // if [src_pos + length > arrayOop(src)->length()] then FAIL
1619     __ lwu(t0, Address(src, arrayOopDesc::length_offset_in_bytes()));
1620     __ addw(temp, length, src_pos);
1621     __ bgtu(temp, t0, L_failed);
1622 
1623     // if [dst_pos + length > arrayOop(dst)->length()] then FAIL
1624     __ lwu(t0, Address(dst, arrayOopDesc::length_offset_in_bytes()));
1625     __ addw(temp, length, dst_pos);
1626     __ bgtu(temp, t0, L_failed);
1627 
1628     // Have to clean up high 32 bits of 'src_pos' and 'dst_pos'.
1629     __ zext(src_pos, src_pos, 32);
1630     __ zext(dst_pos, dst_pos, 32);
1631 
1632     BLOCK_COMMENT("arraycopy_range_checks done");
1633   }
1634 
1635   address generate_unsafecopy_common_error_exit() {
1636     address start = __ pc();
1637     __ mv(x10, 0);
1638     __ leave();
1639     __ ret();
1640     return start;
1641   }
1642 
1643   //
1644   //  Generate 'unsafe' set memory stub
1645   //  Though just as safe as the other stubs, it takes an unscaled
1646   //  size_t (# bytes) argument instead of an element count.
1647   //
1648   //  Input:
1649   //    c_rarg0   - destination array address
1650   //    c_rarg1   - byte count (size_t)
1651   //    c_rarg2   - byte value
1652   //
1653   address generate_unsafe_setmemory() {
1654     __ align(CodeEntryAlignment);
1655     StubId stub_id = StubId::stubgen_unsafe_setmemory_id;
1656     StubCodeMark mark(this, stub_id);
1657     address start = __ pc();
1658 
1659     // bump this on entry, not on exit:
1660     // inc_counter_np(SharedRuntime::_unsafe_set_memory_ctr);
1661 
1662     Label L_fill_elements;
1663 
1664     const Register dest = c_rarg0;
1665     const Register count = c_rarg1;
1666     const Register value = c_rarg2;
1667     const Register cnt_words = x28; // temp register
1668     const Register tmp_reg   = x29; // temp register
1669 
1670     // Mark remaining code as such which performs Unsafe accesses.
1671     UnsafeMemoryAccessMark umam(this, true, false);
1672 
1673     __ enter(); // required for proper stackwalking of RuntimeStub frame
1674 
1675     // if count < 8, jump to L_fill_elements
1676     __ mv(tmp_reg, 8); // 8 bytes fill by element
1677     __ bltu(count, tmp_reg, L_fill_elements);
1678 
1679     // Propagate byte to 64-bit width
1680     // 8 bit -> 16 bit
1681     __ zext(value, value, 8);
1682     __ slli(tmp_reg, value, 8);
1683     __ orr(value, value, tmp_reg);
1684     // 16 bit -> 32 bit
1685     __ slli(tmp_reg, value, 16);
1686     __ orr(value, value, tmp_reg);
1687     // 32 bit -> 64 bit
1688     __ slli(tmp_reg, value, 32);
1689     __ orr(value, value, tmp_reg);
1690 
1691     // Align source address at 8 bytes address boundary.
1692     Label L_skip_align1, L_skip_align2, L_skip_align4;
1693     // One byte misalignment happens.
1694     __ test_bit(tmp_reg, dest, 0);
1695     __ beqz(tmp_reg, L_skip_align1);
1696     __ sb(value, Address(dest, 0));
1697     __ addi(dest, dest, 1);
1698     __ subi(count, count, 1);
1699 
1700     __ bind(L_skip_align1);
1701     // Two bytes misalignment happens.
1702     __ test_bit(tmp_reg, dest, 1);
1703     __ beqz(tmp_reg, L_skip_align2);
1704     __ sh(value, Address(dest, 0));
1705     __ addi(dest, dest, 2);
1706     __ subi(count, count, 2);
1707 
1708     __ bind(L_skip_align2);
1709     // Four bytes misalignment happens.
1710     __ test_bit(tmp_reg, dest, 2);
1711     __ beqz(tmp_reg, L_skip_align4);
1712     __ sw(value, Address(dest, 0));
1713     __ addi(dest, dest, 4);
1714     __ subi(count, count, 4);
1715     __ bind(L_skip_align4);
1716 
1717     //  Fill large chunks
1718     __ srli(cnt_words, count, 3); // number of words
1719     __ slli(tmp_reg, cnt_words, 3);
1720     __ sub(count, count, tmp_reg);
1721     {
1722       __ fill_words(dest, cnt_words, value);
1723     }
1724 
1725     // Handle copies less than 8 bytes
1726     __ bind(L_fill_elements);
1727     Label L_fill_2, L_fill_1, L_exit;
1728     __ test_bit(tmp_reg, count, 2);
1729     __ beqz(tmp_reg, L_fill_2);
1730     __ sb(value, Address(dest, 0));
1731     __ sb(value, Address(dest, 1));
1732     __ sb(value, Address(dest, 2));
1733     __ sb(value, Address(dest, 3));
1734     __ addi(dest, dest, 4);
1735 
1736     __ bind(L_fill_2);
1737     __ test_bit(tmp_reg, count, 1);
1738     __ beqz(tmp_reg, L_fill_1);
1739     __ sb(value, Address(dest, 0));
1740     __ sb(value, Address(dest, 1));
1741     __ addi(dest, dest, 2);
1742 
1743     __ bind(L_fill_1);
1744     __ test_bit(tmp_reg, count, 0);
1745     __ beqz(tmp_reg, L_exit);
1746     __ sb(value, Address(dest, 0));
1747 
1748     __ bind(L_exit);
1749     __ leave();
1750     __ ret();
1751 
1752     return start;
1753   }
1754 
1755   //
1756   //  Generate 'unsafe' array copy stub
1757   //  Though just as safe as the other stubs, it takes an unscaled
1758   //  size_t argument instead of an element count.
1759   //
1760   //  Input:
1761   //    c_rarg0   - source array address
1762   //    c_rarg1   - destination array address
1763   //    c_rarg2   - byte count, treated as ssize_t, can be zero
1764   //
1765   // Examines the alignment of the operands and dispatches
1766   // to a long, int, short, or byte copy loop.
1767   //
1768   address generate_unsafe_copy(address byte_copy_entry,
1769                                address short_copy_entry,
1770                                address int_copy_entry,
1771                                address long_copy_entry) {
1772     assert_cond(byte_copy_entry != nullptr && short_copy_entry != nullptr &&
1773                 int_copy_entry != nullptr && long_copy_entry != nullptr);
1774     Label L_long_aligned, L_int_aligned, L_short_aligned;
1775     const Register s = c_rarg0, d = c_rarg1, count = c_rarg2;
1776 
1777     __ align(CodeEntryAlignment);
1778     StubId stub_id = StubId::stubgen_unsafe_arraycopy_id;
1779     StubCodeMark mark(this, stub_id);
1780     address start = __ pc();
1781     __ enter(); // required for proper stackwalking of RuntimeStub frame
1782 
1783     // bump this on entry, not on exit:
1784     inc_counter_np(SharedRuntime::_unsafe_array_copy_ctr);
1785 
1786     __ orr(t0, s, d);
1787     __ orr(t0, t0, count);
1788 
1789     __ andi(t0, t0, BytesPerLong - 1);
1790     __ beqz(t0, L_long_aligned);
1791     __ andi(t0, t0, BytesPerInt - 1);
1792     __ beqz(t0, L_int_aligned);
1793     __ test_bit(t0, t0, 0);
1794     __ beqz(t0, L_short_aligned);
1795     __ j(RuntimeAddress(byte_copy_entry));
1796 
1797     __ BIND(L_short_aligned);
1798     __ srli(count, count, LogBytesPerShort);  // size => short_count
1799     __ j(RuntimeAddress(short_copy_entry));
1800     __ BIND(L_int_aligned);
1801     __ srli(count, count, LogBytesPerInt);    // size => int_count
1802     __ j(RuntimeAddress(int_copy_entry));
1803     __ BIND(L_long_aligned);
1804     __ srli(count, count, LogBytesPerLong);   // size => long_count
1805     __ j(RuntimeAddress(long_copy_entry));
1806 
1807     return start;
1808   }
1809 
1810   //
1811   //  Generate generic array copy stubs
1812   //
1813   //  Input:
1814   //    c_rarg0    -  src oop
1815   //    c_rarg1    -  src_pos (32-bits)
1816   //    c_rarg2    -  dst oop
1817   //    c_rarg3    -  dst_pos (32-bits)
1818   //    c_rarg4    -  element count (32-bits)
1819   //
1820   //  Output:
1821   //    x10 ==  0  -  success
1822   //    x10 == -1^K - failure, where K is partial transfer count
1823   //
1824   address generate_generic_copy(address byte_copy_entry, address short_copy_entry,
1825                                 address int_copy_entry, address oop_copy_entry,
1826                                 address long_copy_entry, address checkcast_copy_entry) {
1827     assert_cond(byte_copy_entry != nullptr && short_copy_entry != nullptr &&
1828                 int_copy_entry != nullptr && oop_copy_entry != nullptr &&
1829                 long_copy_entry != nullptr && checkcast_copy_entry != nullptr);
1830     Label L_failed, L_failed_0, L_objArray;
1831     Label L_copy_bytes, L_copy_shorts, L_copy_ints, L_copy_longs;
1832 
1833     // Input registers
1834     const Register src        = c_rarg0;  // source array oop
1835     const Register src_pos    = c_rarg1;  // source position
1836     const Register dst        = c_rarg2;  // destination array oop
1837     const Register dst_pos    = c_rarg3;  // destination position
1838     const Register length     = c_rarg4;
1839 
1840     // Registers used as temps
1841     const Register dst_klass = c_rarg5;
1842 
1843     __ align(CodeEntryAlignment);
1844 
1845     StubId stub_id = StubId::stubgen_generic_arraycopy_id;
1846     StubCodeMark mark(this, stub_id);
1847 
1848     address start = __ pc();
1849 
1850     __ enter(); // required for proper stackwalking of RuntimeStub frame
1851 
1852     // bump this on entry, not on exit:
1853     inc_counter_np(SharedRuntime::_generic_array_copy_ctr);
1854 
1855     //-----------------------------------------------------------------------
1856     // Assembler stub will be used for this call to arraycopy
1857     // if the following conditions are met:
1858     //
1859     // (1) src and dst must not be null.
1860     // (2) src_pos must not be negative.
1861     // (3) dst_pos must not be negative.
1862     // (4) length  must not be negative.
1863     // (5) src klass and dst klass should be the same and not null.
1864     // (6) src and dst should be arrays.
1865     // (7) src_pos + length must not exceed length of src.
1866     // (8) dst_pos + length must not exceed length of dst.
1867     //
1868 
1869     // if src is null then return -1
1870     __ beqz(src, L_failed);
1871 
1872     // if [src_pos < 0] then return -1
1873     __ sext(t0, src_pos, 32);
1874     __ bltz(t0, L_failed);
1875 
1876     // if dst is null then return -1
1877     __ beqz(dst, L_failed);
1878 
1879     // if [dst_pos < 0] then return -1
1880     __ sext(t0, dst_pos, 32);
1881     __ bltz(t0, L_failed);
1882 
1883     // registers used as temp
1884     const Register scratch_length    = x28; // elements count to copy
1885     const Register scratch_src_klass = x29; // array klass
1886     const Register lh                = x30; // layout helper
1887 
1888     // if [length < 0] then return -1
1889     __ sext(scratch_length, length, 32); // length (elements count, 32-bits value)
1890     __ bltz(scratch_length, L_failed);
1891 
1892     __ load_narrow_klass(scratch_src_klass, src);
1893 #ifdef ASSERT
1894     {
1895       BLOCK_COMMENT("assert klasses not null {");
1896       Label L1, L2;
1897       __ bnez(scratch_src_klass, L2);   // it is broken if klass is null
1898       __ bind(L1);
1899       __ stop("broken null klass");
1900       __ bind(L2);
1901       __ load_narrow_klass(t0, dst);
1902       __ beqz(t0, L1);     // this would be broken also
1903       BLOCK_COMMENT("} assert klasses not null done");
1904     }
1905 #endif
1906     __ decode_klass_not_null(scratch_src_klass, t0);
1907 
1908     // Load layout helper (32-bits)
1909     //
1910     //  |array_tag|     | header_size | element_type |     |log2_element_size|
1911     // 32        30    24            16              8     2                 0
1912     //
1913     //   array_tag: typeArray = 0x3, objArray = 0x2, non-array = 0x0
1914     //
1915 
1916     const int lh_offset = in_bytes(Klass::layout_helper_offset());
1917 
1918     // Handle objArrays completely differently...
1919     const jint objArray_lh = Klass::array_layout_helper(T_OBJECT);
1920     __ lw(lh, Address(scratch_src_klass, lh_offset));
1921     __ mv(t0, objArray_lh);
1922     __ beq(lh, t0, L_objArray);
1923 
1924     // if [src->klass() != dst->klass()] then return -1
1925     __ load_klass(t1, dst);
1926     __ bne(t1, scratch_src_klass, L_failed);
1927 
1928     // if src->is_Array() isn't null then return -1
1929     // i.e. (lh >= 0)
1930     __ bgez(lh, L_failed);
1931 
1932     // At this point, it is known to be a typeArray (array_tag 0x3).
1933 #ifdef ASSERT
1934     {
1935       BLOCK_COMMENT("assert primitive array {");
1936       Label L;
1937       __ mv(t1, (int32_t)(Klass::_lh_array_tag_type_value << Klass::_lh_array_tag_shift));
1938       __ bge(lh, t1, L);
1939       __ stop("must be a primitive array");
1940       __ bind(L);
1941       BLOCK_COMMENT("} assert primitive array done");
1942     }
1943 #endif
1944 
1945     arraycopy_range_checks(src, src_pos, dst, dst_pos, scratch_length,
1946                            t1, L_failed);
1947 
1948     // TypeArrayKlass
1949     //
1950     // src_addr = (src + array_header_in_bytes()) + (src_pos << log2elemsize)
1951     // dst_addr = (dst + array_header_in_bytes()) + (dst_pos << log2elemsize)
1952     //
1953 
1954     const Register t0_offset = t0;    // array offset
1955     const Register x30_elsize = lh;   // element size
1956 
1957     // Get array_header_in_bytes()
1958     int lh_header_size_width = exact_log2(Klass::_lh_header_size_mask + 1);
1959     int lh_header_size_msb = Klass::_lh_header_size_shift + lh_header_size_width;
1960     __ slli(t0_offset, lh, XLEN - lh_header_size_msb);          // left shift to remove 24 ~ 32;
1961     __ srli(t0_offset, t0_offset, XLEN - lh_header_size_width); // array_offset
1962 
1963     __ add(src, src, t0_offset);           // src array offset
1964     __ add(dst, dst, t0_offset);           // dst array offset
1965     BLOCK_COMMENT("choose copy loop based on element size");
1966 
1967     // next registers should be set before the jump to corresponding stub
1968     const Register from     = c_rarg0;  // source array address
1969     const Register to       = c_rarg1;  // destination array address
1970     const Register count    = c_rarg2;  // elements count
1971 
1972     // 'from', 'to', 'count' registers should be set in such order
1973     // since they are the same as 'src', 'src_pos', 'dst'.
1974 
1975     assert(Klass::_lh_log2_element_size_shift == 0, "fix this code");
1976 
1977     // The possible values of elsize are 0-3, i.e. exact_log2(element
1978     // size in bytes).  We do a simple bitwise binary search.
1979   __ BIND(L_copy_bytes);
1980     __ test_bit(t0, x30_elsize, 1);
1981     __ bnez(t0, L_copy_ints);
1982     __ test_bit(t0, x30_elsize, 0);
1983     __ bnez(t0, L_copy_shorts);
1984     __ add(from, src, src_pos); // src_addr
1985     __ add(to, dst, dst_pos); // dst_addr
1986     __ sext(count, scratch_length, 32); // length
1987     __ j(RuntimeAddress(byte_copy_entry));
1988 
1989   __ BIND(L_copy_shorts);
1990     __ shadd(from, src_pos, src, t0, 1); // src_addr
1991     __ shadd(to, dst_pos, dst, t0, 1); // dst_addr
1992     __ sext(count, scratch_length, 32); // length
1993     __ j(RuntimeAddress(short_copy_entry));
1994 
1995   __ BIND(L_copy_ints);
1996     __ test_bit(t0, x30_elsize, 0);
1997     __ bnez(t0, L_copy_longs);
1998     __ shadd(from, src_pos, src, t0, 2); // src_addr
1999     __ shadd(to, dst_pos, dst, t0, 2); // dst_addr
2000     __ sext(count, scratch_length, 32); // length
2001     __ j(RuntimeAddress(int_copy_entry));
2002 
2003   __ BIND(L_copy_longs);
2004 #ifdef ASSERT
2005     {
2006       BLOCK_COMMENT("assert long copy {");
2007       Label L;
2008       __ andi(lh, lh, Klass::_lh_log2_element_size_mask); // lh -> x30_elsize
2009       __ sext(lh, lh, 32);
2010       __ mv(t0, LogBytesPerLong);
2011       __ beq(x30_elsize, t0, L);
2012       __ stop("must be long copy, but elsize is wrong");
2013       __ bind(L);
2014       BLOCK_COMMENT("} assert long copy done");
2015     }
2016 #endif
2017     __ shadd(from, src_pos, src, t0, 3); // src_addr
2018     __ shadd(to, dst_pos, dst, t0, 3); // dst_addr
2019     __ sext(count, scratch_length, 32); // length
2020     __ j(RuntimeAddress(long_copy_entry));
2021 
2022     // ObjArrayKlass
2023   __ BIND(L_objArray);
2024     // live at this point:  scratch_src_klass, scratch_length, src[_pos], dst[_pos]
2025 
2026     Label L_plain_copy, L_checkcast_copy;
2027     // test array classes for subtyping
2028     __ load_klass(t2, dst);
2029     __ bne(scratch_src_klass, t2, L_checkcast_copy); // usual case is exact equality
2030 
2031     // Identically typed arrays can be copied without element-wise checks.
2032     arraycopy_range_checks(src, src_pos, dst, dst_pos, scratch_length,
2033                            t1, L_failed);
2034 
2035     __ shadd(from, src_pos, src, t0, LogBytesPerHeapOop);
2036     __ addi(from, from, arrayOopDesc::base_offset_in_bytes(T_OBJECT));
2037     __ shadd(to, dst_pos, dst, t0, LogBytesPerHeapOop);
2038     __ addi(to, to, arrayOopDesc::base_offset_in_bytes(T_OBJECT));
2039     __ sext(count, scratch_length, 32); // length
2040   __ BIND(L_plain_copy);
2041     __ j(RuntimeAddress(oop_copy_entry));
2042 
2043   __ BIND(L_checkcast_copy);
2044     // live at this point:  scratch_src_klass, scratch_length, t2 (dst_klass)
2045     {
2046       // Before looking at dst.length, make sure dst is also an objArray.
2047       __ lwu(t0, Address(t2, lh_offset));
2048       __ mv(t1, objArray_lh);
2049       __ bne(t0, t1, L_failed);
2050 
2051       // It is safe to examine both src.length and dst.length.
2052       arraycopy_range_checks(src, src_pos, dst, dst_pos, scratch_length,
2053                              t2, L_failed);
2054 
2055       __ load_klass(dst_klass, dst); // reload
2056 
2057       // Marshal the base address arguments now, freeing registers.
2058       __ shadd(from, src_pos, src, t0, LogBytesPerHeapOop);
2059       __ addi(from, from, arrayOopDesc::base_offset_in_bytes(T_OBJECT));
2060       __ shadd(to, dst_pos, dst, t0, LogBytesPerHeapOop);
2061       __ addi(to, to, arrayOopDesc::base_offset_in_bytes(T_OBJECT));
2062       __ sext(count, length, 32); // length (reloaded)
2063       const Register sco_temp = c_rarg3; // this register is free now
2064       assert_different_registers(from, to, count, sco_temp,
2065                                  dst_klass, scratch_src_klass);
2066 
2067       // Generate the type check.
2068       const int sco_offset = in_bytes(Klass::super_check_offset_offset());
2069       __ lwu(sco_temp, Address(dst_klass, sco_offset));
2070 
2071       // Smashes t0, t1
2072       generate_type_check(scratch_src_klass, sco_temp, dst_klass, noreg, noreg, noreg, L_plain_copy);
2073 
2074       // Fetch destination element klass from the ObjArrayKlass header.
2075       int ek_offset = in_bytes(ObjArrayKlass::element_klass_offset());
2076       __ ld(dst_klass, Address(dst_klass, ek_offset));
2077       __ lwu(sco_temp, Address(dst_klass, sco_offset));
2078 
2079       // the checkcast_copy loop needs two extra arguments:
2080       assert(c_rarg3 == sco_temp, "#3 already in place");
2081       // Set up arguments for checkcast_copy_entry.
2082       __ mv(c_rarg4, dst_klass);  // dst.klass.element_klass
2083       __ j(RuntimeAddress(checkcast_copy_entry));
2084     }
2085 
2086   __ BIND(L_failed);
2087     __ mv(x10, -1);
2088     __ leave();   // required for proper stackwalking of RuntimeStub frame
2089     __ ret();
2090 
2091     return start;
2092   }
2093 
2094   //
2095   // Generate stub for array fill. If "aligned" is true, the
2096   // "to" address is assumed to be heapword aligned.
2097   //
2098   // Arguments for generated stub:
2099   //   to:    c_rarg0
2100   //   value: c_rarg1
2101   //   count: c_rarg2 treated as signed
2102   //
2103   address generate_fill(StubId stub_id) {
2104     BasicType t;
2105     bool aligned;
2106 
2107     switch (stub_id) {
2108     case StubId::stubgen_jbyte_fill_id:
2109       t = T_BYTE;
2110       aligned = false;
2111       break;
2112     case StubId::stubgen_jshort_fill_id:
2113       t = T_SHORT;
2114       aligned = false;
2115       break;
2116     case StubId::stubgen_jint_fill_id:
2117       t = T_INT;
2118       aligned = false;
2119       break;
2120     case StubId::stubgen_arrayof_jbyte_fill_id:
2121       t = T_BYTE;
2122       aligned = true;
2123       break;
2124     case StubId::stubgen_arrayof_jshort_fill_id:
2125       t = T_SHORT;
2126       aligned = true;
2127       break;
2128     case StubId::stubgen_arrayof_jint_fill_id:
2129       t = T_INT;
2130       aligned = true;
2131       break;
2132     default:
2133       ShouldNotReachHere();
2134     };
2135 
2136     __ align(CodeEntryAlignment);
2137     StubCodeMark mark(this, stub_id);
2138     address start = __ pc();
2139 
2140     BLOCK_COMMENT("Entry:");
2141 
2142     const Register to        = c_rarg0;  // source array address
2143     const Register value     = c_rarg1;  // value
2144     const Register count     = c_rarg2;  // elements count
2145 
2146     const Register bz_base   = x28;      // base for block_zero routine
2147     const Register cnt_words = x29;      // temp register
2148     const Register tmp_reg   = t1;
2149 
2150     __ enter();
2151 
2152     Label L_fill_elements;
2153 
2154     int shift = -1;
2155     switch (t) {
2156       case T_BYTE:
2157         shift = 0;
2158         // Short arrays (< 8 bytes) fill by element
2159         __ mv(tmp_reg, 8 >> shift);
2160         __ bltu(count, tmp_reg, L_fill_elements);
2161 
2162         // Zero extend value
2163         // 8 bit -> 16 bit
2164         __ zext(value, value, 8);
2165         __ slli(tmp_reg, value, 8);
2166         __ orr(value, value, tmp_reg);
2167 
2168         // 16 bit -> 32 bit
2169         __ slli(tmp_reg, value, 16);
2170         __ orr(value, value, tmp_reg);
2171         break;
2172       case T_SHORT:
2173         shift = 1;
2174         // Short arrays (< 8 bytes) fill by element
2175         __ mv(tmp_reg, 8 >> shift);
2176         __ bltu(count, tmp_reg, L_fill_elements);
2177 
2178         // Zero extend value
2179         // 16 bit -> 32 bit
2180         __ zext(value, value, 16);
2181         __ slli(tmp_reg, value, 16);
2182         __ orr(value, value, tmp_reg);
2183         break;
2184       case T_INT:
2185         shift = 2;
2186         // Short arrays (< 8 bytes) fill by element
2187         __ mv(tmp_reg, 8 >> shift);
2188         __ bltu(count, tmp_reg, L_fill_elements);
2189         break;
2190       default: ShouldNotReachHere();
2191     }
2192 
2193     // Align source address at 8 bytes address boundary.
2194     Label L_skip_align1, L_skip_align2, L_skip_align4;
2195     if (!aligned) {
2196       switch (t) {
2197         case T_BYTE:
2198           // One byte misalignment happens only for byte arrays.
2199           __ test_bit(tmp_reg, to, 0);
2200           __ beqz(tmp_reg, L_skip_align1);
2201           __ sb(value, Address(to, 0));
2202           __ addi(to, to, 1);
2203           __ subiw(count, count, 1);
2204           __ bind(L_skip_align1);
2205           // Fallthrough
2206         case T_SHORT:
2207           // Two bytes misalignment happens only for byte and short (char) arrays.
2208           __ test_bit(tmp_reg, to, 1);
2209           __ beqz(tmp_reg, L_skip_align2);
2210           __ sh(value, Address(to, 0));
2211           __ addi(to, to, 2);
2212           __ subiw(count, count, 2 >> shift);
2213           __ bind(L_skip_align2);
2214           // Fallthrough
2215         case T_INT:
2216           // Align to 8 bytes, we know we are 4 byte aligned to start.
2217           __ test_bit(tmp_reg, to, 2);
2218           __ beqz(tmp_reg, L_skip_align4);
2219           __ sw(value, Address(to, 0));
2220           __ addi(to, to, 4);
2221           __ subiw(count, count, 4 >> shift);
2222           __ bind(L_skip_align4);
2223           break;
2224         default: ShouldNotReachHere();
2225       }
2226     }
2227 
2228     //
2229     //  Fill large chunks
2230     //
2231     __ srliw(cnt_words, count, 3 - shift); // number of words
2232 
2233     // 32 bit -> 64 bit
2234     __ zext(value, value, 32);
2235     __ slli(tmp_reg, value, 32);
2236     __ orr(value, value, tmp_reg);
2237 
2238     __ slli(tmp_reg, cnt_words, 3 - shift);
2239     __ subw(count, count, tmp_reg);
2240     {
2241       __ fill_words(to, cnt_words, value);
2242     }
2243 
2244     // Handle copies less than 8 bytes.
2245     // Address may not be heapword aligned.
2246     Label L_fill_1, L_fill_2, L_exit;
2247     __ bind(L_fill_elements);
2248     switch (t) {
2249       case T_BYTE:
2250         __ test_bit(tmp_reg, count, 2);
2251         __ beqz(tmp_reg, L_fill_2);
2252         __ sb(value, Address(to, 0));
2253         __ sb(value, Address(to, 1));
2254         __ sb(value, Address(to, 2));
2255         __ sb(value, Address(to, 3));
2256         __ addi(to, to, 4);
2257 
2258         __ bind(L_fill_2);
2259         __ test_bit(tmp_reg, count, 1);
2260         __ beqz(tmp_reg, L_fill_1);
2261         __ sb(value, Address(to, 0));
2262         __ sb(value, Address(to, 1));
2263         __ addi(to, to, 2);
2264 
2265         __ bind(L_fill_1);
2266         __ test_bit(tmp_reg, count, 0);
2267         __ beqz(tmp_reg, L_exit);
2268         __ sb(value, Address(to, 0));
2269         break;
2270       case T_SHORT:
2271         __ test_bit(tmp_reg, count, 1);
2272         __ beqz(tmp_reg, L_fill_2);
2273         __ sh(value, Address(to, 0));
2274         __ sh(value, Address(to, 2));
2275         __ addi(to, to, 4);
2276 
2277         __ bind(L_fill_2);
2278         __ test_bit(tmp_reg, count, 0);
2279         __ beqz(tmp_reg, L_exit);
2280         __ sh(value, Address(to, 0));
2281         break;
2282       case T_INT:
2283         __ beqz(count, L_exit);
2284         __ sw(value, Address(to, 0));
2285         break;
2286       default: ShouldNotReachHere();
2287     }
2288     __ bind(L_exit);
2289     __ leave();
2290     __ ret();
2291 
2292     return start;
2293   }
2294 
2295   void generate_arraycopy_stubs() {
2296     // Some copy stubs publish a normal entry and then a 2nd 'fallback'
2297     // entry immediately following their stack push. This can be used
2298     // as a post-push branch target for compatible stubs when they
2299     // identify a special case that can be handled by the fallback
2300     // stub e.g a disjoint copy stub may be use as a special case
2301     // fallback for its compatible conjoint copy stub.
2302     //
2303     // A no push entry is always returned in the following local and
2304     // then published by assigning to the appropriate entry field in
2305     // class StubRoutines. The entry value is then passed to the
2306     // generator for the compatible stub. That means the entry must be
2307     // listed when saving to/restoring from the AOT cache, ensuring
2308     // that the inter-stub jumps are noted at AOT-cache save and
2309     // relocated at AOT cache load.
2310     address nopush_entry = nullptr;
2311 
2312     // generate the common exit first so later stubs can rely on it if
2313     // they want an UnsafeMemoryAccess exit non-local to the stub
2314     StubRoutines::_unsafecopy_common_exit = generate_unsafecopy_common_error_exit();
2315     // register the stub as the default exit with class UnsafeMemoryAccess
2316     UnsafeMemoryAccess::set_common_exit_stub_pc(StubRoutines::_unsafecopy_common_exit);
2317 
2318     // generate and publish riscv-specific bulk copy routines first
2319     // so we can call them from other copy stubs
2320     StubRoutines::riscv::_copy_byte_f = generate_copy_longs(StubId::stubgen_copy_byte_f_id, c_rarg0, c_rarg1, t1);
2321     StubRoutines::riscv::_copy_byte_b = generate_copy_longs(StubId::stubgen_copy_byte_b_id, c_rarg0, c_rarg1, t1);
2322 
2323     StubRoutines::riscv::_zero_blocks = generate_zero_blocks();
2324 
2325     //*** jbyte
2326     // Always need aligned and unaligned versions
2327     StubRoutines::_jbyte_disjoint_arraycopy          = generate_disjoint_copy(StubId::stubgen_jbyte_disjoint_arraycopy_id, &nopush_entry);
2328     // disjoint nopush entry is needed by conjoint copy
2329     StubRoutines::_jbyte_disjoint_arraycopy_nopush  = nopush_entry;
2330     StubRoutines::_jbyte_arraycopy                   = generate_conjoint_copy(StubId::stubgen_jbyte_arraycopy_id, StubRoutines::_jbyte_disjoint_arraycopy_nopush, &nopush_entry);
2331     // conjoint nopush entry is needed by generic/unsafe copy
2332     StubRoutines::_jbyte_arraycopy_nopush = nopush_entry;
2333     StubRoutines::_arrayof_jbyte_disjoint_arraycopy  = generate_disjoint_copy(StubId::stubgen_arrayof_jbyte_disjoint_arraycopy_id, &nopush_entry);
2334     // disjoint arrayof nopush entry is needed by conjoint copy
2335     StubRoutines::_arrayof_jbyte_disjoint_arraycopy_nopush  = nopush_entry;
2336     StubRoutines::_arrayof_jbyte_arraycopy           = generate_conjoint_copy(StubId::stubgen_arrayof_jbyte_arraycopy_id, StubRoutines::_arrayof_jbyte_disjoint_arraycopy_nopush, nullptr);
2337 
2338     //*** jshort
2339     // Always need aligned and unaligned versions
2340     StubRoutines::_jshort_disjoint_arraycopy         = generate_disjoint_copy(StubId::stubgen_jshort_disjoint_arraycopy_id, &nopush_entry);
2341     // disjoint nopush entry is needed by conjoint copy
2342     StubRoutines::_jshort_disjoint_arraycopy_nopush  = nopush_entry;
2343     StubRoutines::_jshort_arraycopy                  = generate_conjoint_copy(StubId::stubgen_jshort_arraycopy_id, StubRoutines::_jshort_disjoint_arraycopy_nopush, &nopush_entry);
2344     // conjoint nopush entry is used by generic/unsafe copy
2345     StubRoutines::_jshort_arraycopy_nopush = nopush_entry;
2346     StubRoutines::_arrayof_jshort_disjoint_arraycopy = generate_disjoint_copy(StubId::stubgen_arrayof_jshort_disjoint_arraycopy_id, &nopush_entry);
2347     // disjoint arrayof nopush entry is needed by conjoint copy
2348     StubRoutines::_arrayof_jshort_disjoint_arraycopy_nopush = nopush_entry;
2349     StubRoutines::_arrayof_jshort_arraycopy          = generate_conjoint_copy(StubId::stubgen_arrayof_jshort_arraycopy_id, StubRoutines::_arrayof_jshort_disjoint_arraycopy_nopush, nullptr);
2350 
2351     //*** jint
2352     // Aligned versions
2353     StubRoutines::_arrayof_jint_disjoint_arraycopy   = generate_disjoint_copy(StubId::stubgen_arrayof_jint_disjoint_arraycopy_id, &nopush_entry);
2354     // disjoint arrayof nopush entry is needed by conjoint copy
2355     StubRoutines::_arrayof_jint_disjoint_arraycopy_nopush = nopush_entry;
2356     StubRoutines::_arrayof_jint_arraycopy            = generate_conjoint_copy(StubId::stubgen_arrayof_jint_arraycopy_id, StubRoutines::_arrayof_jint_disjoint_arraycopy_nopush, nullptr);
2357     // In 64 bit we need both aligned and unaligned versions of jint arraycopy.
2358     // entry_jint_arraycopy always points to the unaligned version
2359     StubRoutines::_jint_disjoint_arraycopy           = generate_disjoint_copy(StubId::stubgen_jint_disjoint_arraycopy_id, &nopush_entry);
2360     // disjoint nopush entry is needed by conjoint copy
2361     StubRoutines::_jint_disjoint_arraycopy_nopush  = nopush_entry;
2362     StubRoutines::_jint_arraycopy                  = generate_conjoint_copy(StubId::stubgen_jint_arraycopy_id, StubRoutines::_jint_disjoint_arraycopy_nopush, &nopush_entry);
2363     // conjoint nopush entry is needed by generic/unsafe copy
2364     StubRoutines::_jint_arraycopy_nopush = nopush_entry;
2365 
2366     //*** jlong
2367     // It is always aligned
2368     StubRoutines::_arrayof_jlong_disjoint_arraycopy = generate_disjoint_copy(StubId::stubgen_arrayof_jlong_disjoint_arraycopy_id, &nopush_entry);
2369     // disjoint arrayof nopush entry is needed by conjoint copy
2370     StubRoutines::_arrayof_jlong_disjoint_arraycopy_nopush = nopush_entry;
2371     StubRoutines::_arrayof_jlong_arraycopy          = generate_conjoint_copy(StubId::stubgen_arrayof_jlong_arraycopy_id, StubRoutines::_arrayof_jlong_disjoint_arraycopy_nopush, &nopush_entry);
2372     // conjoint nopush entry is needed by generic/unsafe copy
2373     StubRoutines::_jlong_arraycopy_nopush = nopush_entry;
2374     // disjoint normal/nopush and conjoint normal entries are not
2375     // generated since the arrayof versions are the same
2376     StubRoutines::_jlong_disjoint_arraycopy         = StubRoutines::_arrayof_jlong_disjoint_arraycopy;
2377     StubRoutines::_jlong_disjoint_arraycopy_nopush = StubRoutines::_arrayof_jlong_disjoint_arraycopy_nopush;
2378     StubRoutines::_jlong_arraycopy                  = StubRoutines::_arrayof_jlong_arraycopy;
2379 
2380     //*** oops
2381     StubRoutines::_arrayof_oop_disjoint_arraycopy
2382       = generate_disjoint_copy(StubId::stubgen_arrayof_oop_disjoint_arraycopy_id, &nopush_entry);
2383       // disjoint arrayof nopush entry is needed by conjoint copy
2384     StubRoutines::_arrayof_oop_disjoint_arraycopy_nopush = nopush_entry;
2385     StubRoutines::_arrayof_oop_arraycopy
2386       = generate_conjoint_copy(StubId::stubgen_arrayof_oop_arraycopy_id, StubRoutines::_arrayof_oop_disjoint_arraycopy_nopush, &nopush_entry);
2387     // conjoint arrayof nopush entry is needed by generic/unsafe copy
2388     StubRoutines::_oop_arraycopy_nopush = nopush_entry;
2389     // Aligned versions without pre-barriers
2390     StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit
2391       = generate_disjoint_copy(StubId::stubgen_arrayof_oop_disjoint_arraycopy_uninit_id, &nopush_entry);
2392     // disjoint arrayof+uninit nopush entry is needed by conjoint copy
2393     StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit_nopush = nopush_entry;
2394 
2395     // note that we don't need a returned nopush entry because the
2396     // generic/unsafe copy does not cater for uninit arrays.
2397     StubRoutines::_arrayof_oop_arraycopy_uninit
2398       = generate_conjoint_copy(StubId::stubgen_arrayof_oop_arraycopy_uninit_id, StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit_nopush, nullptr);
2399 
2400     // for oop copies reuse arrayof entries for non-arrayof cases
2401     StubRoutines::_oop_disjoint_arraycopy            = StubRoutines::_arrayof_oop_disjoint_arraycopy;
2402     StubRoutines::_oop_disjoint_arraycopy_nopush = StubRoutines::_arrayof_oop_disjoint_arraycopy_nopush;
2403     StubRoutines::_oop_arraycopy                     = StubRoutines::_arrayof_oop_arraycopy;
2404     StubRoutines::_oop_disjoint_arraycopy_uninit     = StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit;
2405     StubRoutines::_oop_disjoint_arraycopy_uninit_nopush = StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit_nopush;
2406     StubRoutines::_oop_arraycopy_uninit              = StubRoutines::_arrayof_oop_arraycopy_uninit;
2407 
2408     StubRoutines::_checkcast_arraycopy        = generate_checkcast_copy(StubId::stubgen_checkcast_arraycopy_id, &nopush_entry);
2409     // checkcast nopush entry is needed by generic copy
2410     StubRoutines::_checkcast_arraycopy_nopush = nopush_entry;
2411     // note that we don't need a returned nopush entry because the
2412     // generic copy does not cater for uninit arrays.
2413     StubRoutines::_checkcast_arraycopy_uninit = generate_checkcast_copy(StubId::stubgen_checkcast_arraycopy_uninit_id, nullptr);
2414 
2415 
2416     // unsafe arraycopy may fallback on conjoint stubs
2417     StubRoutines::_unsafe_arraycopy    = generate_unsafe_copy(StubRoutines::_jbyte_arraycopy_nopush,
2418                                                               StubRoutines::_jshort_arraycopy_nopush,
2419                                                               StubRoutines::_jint_arraycopy_nopush,
2420                                                               StubRoutines::_jlong_arraycopy_nopush);
2421 
2422     // generic arraycopy may fallback on conjoint stubs
2423     StubRoutines::_generic_arraycopy   = generate_generic_copy(StubRoutines::_jbyte_arraycopy_nopush,
2424                                                                StubRoutines::_jshort_arraycopy_nopush,
2425                                                                StubRoutines::_jint_arraycopy_nopush,
2426                                                                StubRoutines::_oop_arraycopy_nopush,
2427                                                                StubRoutines::_jlong_arraycopy_nopush,
2428                                                                StubRoutines::_checkcast_arraycopy_nopush);
2429 
2430     StubRoutines::_jbyte_fill = generate_fill(StubId::stubgen_jbyte_fill_id);
2431     StubRoutines::_jshort_fill = generate_fill(StubId::stubgen_jshort_fill_id);
2432     StubRoutines::_jint_fill = generate_fill(StubId::stubgen_jint_fill_id);
2433     StubRoutines::_arrayof_jbyte_fill = generate_fill(StubId::stubgen_arrayof_jbyte_fill_id);
2434     StubRoutines::_arrayof_jshort_fill = generate_fill(StubId::stubgen_arrayof_jshort_fill_id);
2435     StubRoutines::_arrayof_jint_fill = generate_fill(StubId::stubgen_arrayof_jint_fill_id);
2436 
2437     StubRoutines::_unsafe_setmemory    = generate_unsafe_setmemory();
2438   }
2439 
2440   void aes_load_keys(const Register &key, VectorRegister *working_vregs, int rounds) {
2441     const int step = 16;
2442     for (int i = 0; i < rounds; i++) {
2443       __ vle32_v(working_vregs[i], key);
2444       // The keys are stored in little-endian array, while we need
2445       // to operate in big-endian.
2446       // So performing an endian-swap here with vrev8.v instruction
2447       __ vrev8_v(working_vregs[i], working_vregs[i]);
2448       __ addi(key, key, step);
2449     }
2450   }
2451 
2452   void aes_encrypt(const VectorRegister &res, VectorRegister *working_vregs, int rounds) {
2453     assert(rounds <= 15, "rounds should be less than or equal to working_vregs size");
2454 
2455     __ vxor_vv(res, res, working_vregs[0]);
2456     for (int i = 1; i < rounds - 1; i++) {
2457       __ vaesem_vv(res, working_vregs[i]);
2458     }
2459     __ vaesef_vv(res, working_vregs[rounds - 1]);
2460   }
2461 
2462   // Arguments:
2463   //
2464   // Inputs:
2465   //   c_rarg0   - source byte array address
2466   //   c_rarg1   - destination byte array address
2467   //   c_rarg2   - sessionKe (key) in little endian int array
2468   //
2469   address generate_aescrypt_encryptBlock() {
2470     assert(UseAESIntrinsics, "need AES instructions (Zvkned extension) support");
2471 
2472     __ align(CodeEntryAlignment);
2473     StubId stub_id = StubId::stubgen_aescrypt_encryptBlock_id;
2474     StubCodeMark mark(this, stub_id);
2475 
2476     Label L_aes128, L_aes192;
2477 
2478     const Register from        = c_rarg0;  // source array address
2479     const Register to          = c_rarg1;  // destination array address
2480     const Register key         = c_rarg2;  // key array address
2481     const Register keylen      = c_rarg3;
2482 
2483     VectorRegister working_vregs[] = {
2484       v4, v5, v6, v7, v8, v9, v10, v11,
2485       v12, v13, v14, v15, v16, v17, v18
2486     };
2487     const VectorRegister res   = v19;
2488 
2489     address start = __ pc();
2490     __ enter();
2491 
2492     __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2493 
2494     __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2495     __ vle32_v(res, from);
2496 
2497     __ mv(t2, 52); // key length could be only {11, 13, 15} * 4 = {44, 52, 60}
2498     __ bltu(keylen, t2, L_aes128);
2499     __ beq(keylen, t2, L_aes192);
2500     // Else we fallthrough to the biggest case (256-bit key size)
2501 
2502     // Note: the following function performs key += 15*16
2503     aes_load_keys(key, working_vregs, 15);
2504     aes_encrypt(res, working_vregs, 15);
2505     __ vse32_v(res, to);
2506     __ mv(c_rarg0, 0);
2507     __ leave();
2508     __ ret();
2509 
2510   __ bind(L_aes192);
2511     // Note: the following function performs key += 13*16
2512     aes_load_keys(key, working_vregs, 13);
2513     aes_encrypt(res, working_vregs, 13);
2514     __ vse32_v(res, to);
2515     __ mv(c_rarg0, 0);
2516     __ leave();
2517     __ ret();
2518 
2519   __ bind(L_aes128);
2520     // Note: the following function performs key += 11*16
2521     aes_load_keys(key, working_vregs, 11);
2522     aes_encrypt(res, working_vregs, 11);
2523     __ vse32_v(res, to);
2524     __ mv(c_rarg0, 0);
2525     __ leave();
2526     __ ret();
2527 
2528     return start;
2529   }
2530 
2531   void aes_decrypt(const VectorRegister &res, VectorRegister *working_vregs, int rounds) {
2532     assert(rounds <= 15, "rounds should be less than or equal to working_vregs size");
2533 
2534     __ vxor_vv(res, res, working_vregs[rounds - 1]);
2535     for (int i = rounds - 2; i > 0; i--) {
2536       __ vaesdm_vv(res, working_vregs[i]);
2537     }
2538     __ vaesdf_vv(res, working_vregs[0]);
2539   }
2540 
2541   // Arguments:
2542   //
2543   // Inputs:
2544   //   c_rarg0   - source byte array address
2545   //   c_rarg1   - destination byte array address
2546   //   c_rarg2   - sessionKe (key) in little endian int array
2547   //
2548   address generate_aescrypt_decryptBlock() {
2549     assert(UseAESIntrinsics, "need AES instructions (Zvkned extension) support");
2550 
2551     __ align(CodeEntryAlignment);
2552     StubId stub_id = StubId::stubgen_aescrypt_decryptBlock_id;
2553     StubCodeMark mark(this, stub_id);
2554 
2555     Label L_aes128, L_aes192;
2556 
2557     const Register from        = c_rarg0;  // source array address
2558     const Register to          = c_rarg1;  // destination array address
2559     const Register key         = c_rarg2;  // key array address
2560     const Register keylen      = c_rarg3;
2561 
2562     VectorRegister working_vregs[] = {
2563       v4, v5, v6, v7, v8, v9, v10, v11,
2564       v12, v13, v14, v15, v16, v17, v18
2565     };
2566     const VectorRegister res   = v19;
2567 
2568     address start = __ pc();
2569     __ enter(); // required for proper stackwalking of RuntimeStub frame
2570 
2571     __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2572 
2573     __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2574     __ vle32_v(res, from);
2575 
2576     __ mv(t2, 52); // key length could be only {11, 13, 15} * 4 = {44, 52, 60}
2577     __ bltu(keylen, t2, L_aes128);
2578     __ beq(keylen, t2, L_aes192);
2579     // Else we fallthrough to the biggest case (256-bit key size)
2580 
2581     // Note: the following function performs key += 15*16
2582     aes_load_keys(key, working_vregs, 15);
2583     aes_decrypt(res, working_vregs, 15);
2584     __ vse32_v(res, to);
2585     __ mv(c_rarg0, 0);
2586     __ leave();
2587     __ ret();
2588 
2589   __ bind(L_aes192);
2590     // Note: the following function performs key += 13*16
2591     aes_load_keys(key, working_vregs, 13);
2592     aes_decrypt(res, working_vregs, 13);
2593     __ vse32_v(res, to);
2594     __ mv(c_rarg0, 0);
2595     __ leave();
2596     __ ret();
2597 
2598   __ bind(L_aes128);
2599     // Note: the following function performs key += 11*16
2600     aes_load_keys(key, working_vregs, 11);
2601     aes_decrypt(res, working_vregs, 11);
2602     __ vse32_v(res, to);
2603     __ mv(c_rarg0, 0);
2604     __ leave();
2605     __ ret();
2606 
2607     return start;
2608   }
2609 
2610   void cipherBlockChaining_encryptAESCrypt(int round, Register from, Register to, Register key,
2611                                            Register rvec, Register input_len) {
2612     const Register len = x29;
2613 
2614     VectorRegister working_vregs[] = {
2615       v1, v2, v3, v4, v5, v6, v7, v8,
2616       v9, v10, v11, v12, v13, v14, v15
2617     };
2618 
2619     const unsigned int BLOCK_SIZE = 16;
2620 
2621     __ mv(len, input_len);
2622     // load init rvec
2623     __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2624     __ vle32_v(v16, rvec);
2625 
2626     aes_load_keys(key, working_vregs, round);
2627     Label L_enc_loop;
2628     __ bind(L_enc_loop);
2629     // Encrypt from source by block size
2630       __ vle32_v(v17, from);
2631       __ addi(from, from, BLOCK_SIZE);
2632       __ vxor_vv(v16, v16, v17);
2633       aes_encrypt(v16, working_vregs, round);
2634       __ vse32_v(v16, to);
2635       __ addi(to, to, BLOCK_SIZE);
2636       __ subi(len, len, BLOCK_SIZE);
2637       __ bnez(len, L_enc_loop);
2638 
2639     // save current rvec and return
2640     __ vse32_v(v16, rvec);
2641     __ mv(x10, input_len);
2642     __ leave();
2643     __ ret();
2644   }
2645 
2646   // Arguments:
2647   //
2648   // Inputs:
2649   //   c_rarg0   - source byte array address
2650   //   c_rarg1   - destination byte array address
2651   //   c_rarg2   - K (key) in little endian int array
2652   //   c_rarg3   - r vector byte array address
2653   //   c_rarg4   - input length
2654   //
2655   // Output:
2656   //   x10       - input length
2657   //
2658   address generate_cipherBlockChaining_encryptAESCrypt() {
2659     assert(UseAESIntrinsics, "need AES instructions (Zvkned extension) support");
2660     __ align(CodeEntryAlignment);
2661     StubId stub_id = StubId::stubgen_cipherBlockChaining_encryptAESCrypt_id;
2662     StubCodeMark mark(this, stub_id);
2663 
2664     const Register from       = c_rarg0;
2665     const Register to         = c_rarg1;
2666     const Register key        = c_rarg2;
2667     const Register rvec       = c_rarg3;
2668     const Register input_len  = c_rarg4;
2669 
2670     const Register keylen     = x28;
2671 
2672     address start = __ pc();
2673     __ enter();
2674 
2675     Label L_aes128, L_aes192;
2676     // Compute #rounds for AES based on the length of the key array
2677     __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2678     __ mv(t0, 52);
2679     __ bltu(keylen, t0, L_aes128);
2680     __ beq(keylen, t0, L_aes192);
2681     // Else we fallthrough to the biggest case (256-bit key size)
2682 
2683     // Note: the following function performs key += 15*16
2684     cipherBlockChaining_encryptAESCrypt(15, from, to, key, rvec, input_len);
2685 
2686     // Note: the following function performs key += 11*16
2687     __ bind(L_aes128);
2688     cipherBlockChaining_encryptAESCrypt(11, from, to, key, rvec, input_len);
2689 
2690     // Note: the following function performs key += 13*16
2691     __ bind(L_aes192);
2692     cipherBlockChaining_encryptAESCrypt(13, from, to, key, rvec, input_len);
2693 
2694     return start;
2695   }
2696 
2697   void cipherBlockChaining_decryptAESCrypt(int round, Register from, Register to, Register key,
2698                                            Register rvec, Register input_len) {
2699     const Register len = x29;
2700 
2701     VectorRegister working_vregs[] = {
2702       v1, v2, v3, v4, v5, v6, v7, v8,
2703       v9, v10, v11, v12, v13, v14, v15
2704     };
2705 
2706     const unsigned int BLOCK_SIZE = 16;
2707 
2708     __ mv(len, input_len);
2709     // load init rvec
2710     __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2711     __ vle32_v(v16, rvec);
2712 
2713     aes_load_keys(key, working_vregs, round);
2714     Label L_dec_loop;
2715     // Decrypt from source by block size
2716     __ bind(L_dec_loop);
2717       __ vle32_v(v17, from);
2718       __ addi(from, from, BLOCK_SIZE);
2719       __ vmv_v_v(v18, v17);
2720       aes_decrypt(v17, working_vregs, round);
2721       __ vxor_vv(v17, v17, v16);
2722       __ vse32_v(v17, to);
2723       __ vmv_v_v(v16, v18);
2724       __ addi(to, to, BLOCK_SIZE);
2725       __ subi(len, len, BLOCK_SIZE);
2726       __ bnez(len, L_dec_loop);
2727 
2728     // save current rvec and return
2729     __ vse32_v(v16, rvec);
2730     __ mv(x10, input_len);
2731     __ leave();
2732     __ ret();
2733   }
2734 
2735   // Arguments:
2736   //
2737   // Inputs:
2738   //   c_rarg0   - source byte array address
2739   //   c_rarg1   - destination byte array address
2740   //   c_rarg2   - K (key) in little endian int array
2741   //   c_rarg3   - r vector byte array address
2742   //   c_rarg4   - input length
2743   //
2744   // Output:
2745   //   x10       - input length
2746   //
2747   address generate_cipherBlockChaining_decryptAESCrypt() {
2748     assert(UseAESIntrinsics, "need AES instructions (Zvkned extension) support");
2749     __ align(CodeEntryAlignment);
2750     StubId stub_id = StubId::stubgen_cipherBlockChaining_decryptAESCrypt_id;
2751     StubCodeMark mark(this, stub_id);
2752 
2753     const Register from        = c_rarg0;
2754     const Register to          = c_rarg1;
2755     const Register key         = c_rarg2;
2756     const Register rvec        = c_rarg3;
2757     const Register input_len   = c_rarg4;
2758 
2759     const Register keylen      = x28;
2760 
2761     address start = __ pc();
2762     __ enter();
2763 
2764     Label L_aes128, L_aes192, L_aes128_loop, L_aes192_loop, L_aes256_loop;
2765     // Compute #rounds for AES based on the length of the key array
2766     __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2767     __ mv(t0, 52);
2768     __ bltu(keylen, t0, L_aes128);
2769     __ beq(keylen, t0, L_aes192);
2770     // Else we fallthrough to the biggest case (256-bit key size)
2771 
2772     // Note: the following function performs key += 15*16
2773     cipherBlockChaining_decryptAESCrypt(15, from, to, key, rvec, input_len);
2774 
2775     // Note: the following function performs key += 11*16
2776     __ bind(L_aes128);
2777     cipherBlockChaining_decryptAESCrypt(11, from, to, key, rvec, input_len);
2778 
2779     // Note: the following function performs key += 13*16
2780     __ bind(L_aes192);
2781     cipherBlockChaining_decryptAESCrypt(13, from, to, key, rvec, input_len);
2782 
2783     return start;
2784   }
2785 
2786   // Load big-endian 128-bit from memory.
2787   void be_load_counter_128(Register counter_hi, Register counter_lo, Register counter) {
2788     __ ld(counter_lo, Address(counter, 8)); // Load 128-bits from counter
2789     __ ld(counter_hi, Address(counter));
2790     __ rev8(counter_lo, counter_lo);        // Convert big-endian to little-endian
2791     __ rev8(counter_hi, counter_hi);
2792   }
2793 
2794   // Little-endian 128-bit + 64-bit -> 128-bit addition.
2795   void add_counter_128(Register counter_hi, Register counter_lo) {
2796     assert_different_registers(counter_hi, counter_lo, t0);
2797     __ addi(counter_lo, counter_lo, 1);
2798     __ seqz(t0, counter_lo);                // Check for result overflow
2799     __ add(counter_hi, counter_hi, t0);     // Add 1 if overflow otherwise 0
2800   }
2801 
2802   // Store big-endian 128-bit to memory.
2803   void be_store_counter_128(Register counter_hi, Register counter_lo, Register counter) {
2804     assert_different_registers(counter_hi, counter_lo, t0, t1);
2805     __ rev8(t0, counter_lo);                // Convert little-endian to big-endian
2806     __ rev8(t1, counter_hi);
2807     __ sd(t0, Address(counter, 8));         // Store 128-bits to counter
2808     __ sd(t1, Address(counter));
2809   }
2810 
2811   void counterMode_AESCrypt(int round, Register in, Register out, Register key, Register counter,
2812                             Register input_len,  Register saved_encrypted_ctr, Register used_ptr) {
2813     // Algorithm:
2814     //
2815     //   aes_load_keys();
2816     //   load_counter_128(counter_hi, counter_lo, counter);
2817     //
2818     //   L_next:
2819     //     if (used >= BLOCK_SIZE) goto L_main_loop;
2820     //
2821     //   L_encrypt_next:
2822     //       *out = *in ^ saved_encrypted_ctr[used]);
2823     //       out++; in++; used++; len--;
2824     //       if (len == 0) goto L_exit;
2825     //       goto L_next;
2826     //
2827     //   L_main_loop:
2828     //     if (len == 0) goto L_exit;
2829     //     saved_encrypted_ctr = aes_encrypt(counter);
2830     //
2831     //     add_counter_128(counter_hi, counter_lo);
2832     //     be_store_counter_128(counter_hi, counter_lo, counter);
2833     //     used = 0;
2834     //
2835     //     if(len < BLOCK_SIZE) goto L_encrypt_next;
2836     //
2837     //     v_in = load_16Byte(in);
2838     //     v_out = load_16Byte(out);
2839     //     v_saved_encrypted_ctr = load_16Byte(saved_encrypted_ctr);
2840     //     v_out = v_in ^ v_saved_encrypted_ctr;
2841     //     out += BLOCK_SIZE;
2842     //     in += BLOCK_SIZE;
2843     //     len -= BLOCK_SIZE;
2844     //     used = BLOCK_SIZE;
2845     //     goto L_main_loop;
2846     //
2847     //
2848     //   L_exit:
2849     //     store(used);
2850     //     result = input_len
2851     //     return result;
2852 
2853     const Register used          = x28;
2854     const Register len           = x29;
2855     const Register counter_hi    = x30;
2856     const Register counter_lo    = x31;
2857     const Register block_size    = t2;
2858 
2859     const unsigned int BLOCK_SIZE = 16;
2860 
2861     VectorRegister working_vregs[] = {
2862       v1, v2, v3, v4, v5, v6, v7, v8,
2863       v9, v10, v11, v12, v13, v14, v15
2864     };
2865 
2866     __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2867 
2868     __ lwu(used, Address(used_ptr));
2869     __ mv(len, input_len);
2870     __ mv(block_size, BLOCK_SIZE);
2871 
2872     // load keys to working_vregs according to round
2873     aes_load_keys(key, working_vregs, round);
2874 
2875     // 128-bit big-endian load
2876     be_load_counter_128(counter_hi, counter_lo, counter);
2877 
2878     Label L_next, L_encrypt_next, L_main_loop, L_exit;
2879     // Check the last saved_encrypted_ctr used value, we fall through
2880     // to L_encrypt_next when the used value lower than block_size
2881     __ bind(L_next);
2882     __ bgeu(used, block_size, L_main_loop);
2883 
2884     // There is still data left fewer than block_size after L_main_loop
2885     // or last used, we encrypt them one by one.
2886     __ bind(L_encrypt_next);
2887     __ add(t0, saved_encrypted_ctr, used);
2888     __ lbu(t1, Address(t0));
2889     __ lbu(t0, Address(in));
2890     __ xorr(t1, t1, t0);
2891     __ sb(t1, Address(out));
2892     __ addi(in, in, 1);
2893     __ addi(out, out, 1);
2894     __ addi(used, used, 1);
2895     __ subi(len, len, 1);
2896     __ beqz(len, L_exit);
2897     __ j(L_next);
2898 
2899     // We will calculate the next saved_encrypted_ctr and encrypt the blocks of data
2900     // one by one until there is less than a full block remaining if len not zero
2901     __ bind(L_main_loop);
2902     __ beqz(len, L_exit);
2903     __ vle32_v(v16, counter);
2904 
2905     // encrypt counter according to round
2906     aes_encrypt(v16, working_vregs, round);
2907 
2908     __ vse32_v(v16, saved_encrypted_ctr);
2909 
2910     // 128-bit little-endian increment
2911     add_counter_128(counter_hi, counter_lo);
2912     // 128-bit big-endian store
2913     be_store_counter_128(counter_hi, counter_lo, counter);
2914 
2915     __ mv(used, 0);
2916     // Check if we have a full block_size
2917     __ bltu(len, block_size, L_encrypt_next);
2918 
2919     // We have one full block to encrypt at least
2920     __ vle32_v(v17, in);
2921     __ vxor_vv(v16, v16, v17);
2922     __ vse32_v(v16, out);
2923     __ add(out, out, block_size);
2924     __ add(in, in, block_size);
2925     __ sub(len, len, block_size);
2926     __ mv(used, block_size);
2927     __ j(L_main_loop);
2928 
2929     __ bind(L_exit);
2930     __ sw(used, Address(used_ptr));
2931     __ mv(x10, input_len);
2932     __ leave();
2933     __ ret();
2934   };
2935 
2936   // CTR AES crypt.
2937   // Arguments:
2938   //
2939   // Inputs:
2940   //   c_rarg0   - source byte array address
2941   //   c_rarg1   - destination byte array address
2942   //   c_rarg2   - K (key) in little endian int array
2943   //   c_rarg3   - counter vector byte array address
2944   //   c_rarg4   - input length
2945   //   c_rarg5   - saved encryptedCounter start
2946   //   c_rarg6   - saved used length
2947   //
2948   // Output:
2949   //   x10       - input length
2950   //
2951   address generate_counterMode_AESCrypt() {
2952     assert(UseAESCTRIntrinsics, "need AES instructions (Zvkned extension) and Zbb extension support");
2953 
2954     __ align(CodeEntryAlignment);
2955     StubId stub_id = StubId::stubgen_counterMode_AESCrypt_id;
2956     StubCodeMark mark(this, stub_id);
2957 
2958     const Register in                  = c_rarg0;
2959     const Register out                 = c_rarg1;
2960     const Register key                 = c_rarg2;
2961     const Register counter             = c_rarg3;
2962     const Register input_len           = c_rarg4;
2963     const Register saved_encrypted_ctr = c_rarg5;
2964     const Register used_len_ptr        = c_rarg6;
2965 
2966     const Register keylen              = c_rarg7; // temporary register
2967 
2968     const address start = __ pc();
2969     __ enter();
2970 
2971     Label L_exit;
2972     __ beqz(input_len, L_exit);
2973 
2974     Label L_aes128, L_aes192;
2975     // Compute #rounds for AES based on the length of the key array
2976     __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2977     __ mv(t0, 52); // key length could be only {11, 13, 15} * 4 = {44, 52, 60}
2978     __ bltu(keylen, t0, L_aes128);
2979     __ beq(keylen, t0, L_aes192);
2980     // Else we fallthrough to the biggest case (256-bit key size)
2981 
2982     // Note: the following function performs crypt with key += 15*16
2983     counterMode_AESCrypt(15, in, out, key, counter, input_len, saved_encrypted_ctr, used_len_ptr);
2984 
2985     // Note: the following function performs crypt with key += 13*16
2986     __ bind(L_aes192);
2987     counterMode_AESCrypt(13, in, out, key, counter, input_len, saved_encrypted_ctr, used_len_ptr);
2988 
2989     // Note: the following function performs crypt with key += 11*16
2990     __ bind(L_aes128);
2991     counterMode_AESCrypt(11, in, out, key, counter, input_len, saved_encrypted_ctr, used_len_ptr);
2992 
2993     __ bind(L_exit);
2994     __ mv(x10, input_len);
2995     __ leave();
2996     __ ret();
2997 
2998     return start;
2999   }
3000 
3001   void ghash_loop(Register state, Register subkeyH, Register data, Register blocks,
3002                   VectorRegister vtmp1, VectorRegister vtmp2, VectorRegister vtmp3) {
3003     VectorRegister partial_hash = vtmp1;
3004     VectorRegister hash_subkey  = vtmp2;
3005     VectorRegister cipher_text  = vtmp3;
3006 
3007     const unsigned int BLOCK_SIZE = 16;
3008 
3009     __ vsetivli(x0, 2, Assembler::e64, Assembler::m1);
3010     __ vle64_v(hash_subkey, subkeyH);
3011     __ vrev8_v(hash_subkey, hash_subkey);
3012     __ vle64_v(partial_hash, state);
3013     __ vrev8_v(partial_hash, partial_hash);
3014 
3015     __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
3016     Label L_ghash_loop;
3017     __ bind(L_ghash_loop);
3018       __ vle32_v(cipher_text, data);
3019       __ addi(data, data, BLOCK_SIZE);
3020       __ vghsh_vv(partial_hash, hash_subkey, cipher_text);
3021       __ subi(blocks, blocks, 1);
3022       __ bnez(blocks, L_ghash_loop);
3023 
3024     __ vsetivli(x0, 2, Assembler::e64, Assembler::m1);
3025     __ vrev8_v(partial_hash, partial_hash);
3026     __ vse64_v(partial_hash, state);
3027   }
3028 
3029   /**
3030    *  Arguments:
3031    *
3032    *  Input:
3033    *  c_rarg0   - current state address
3034    *  c_rarg1   - H key address
3035    *  c_rarg2   - data address
3036    *  c_rarg3   - number of blocks
3037    *
3038    *  Output:
3039    *  Updated state at c_rarg0
3040    */
3041   address generate_ghash_processBlocks() {
3042     assert(UseGHASHIntrinsics, "need GHASH instructions (Zvkg extension) and Zvbb support");
3043 
3044     __ align(CodeEntryAlignment);
3045     StubId stub_id = StubId::stubgen_ghash_processBlocks_id;
3046     StubCodeMark mark(this, stub_id);
3047 
3048     address start = __ pc();
3049     __ enter();
3050 
3051     Register state   = c_rarg0;
3052     Register subkeyH = c_rarg1;
3053     Register data    = c_rarg2;
3054     Register blocks  = c_rarg3;
3055 
3056     VectorRegister vtmp1 = v1;
3057     VectorRegister vtmp2 = v2;
3058     VectorRegister vtmp3 = v3;
3059 
3060     ghash_loop(state, subkeyH, data, blocks, vtmp1, vtmp2, vtmp3);
3061 
3062     __ leave();
3063     __ ret();
3064 
3065     return start;
3066   }
3067 
3068   void gcm_counterMode_AESCrypt_blocks(int round, Register in, Register out, Register key, Register counter,
3069                                        Register input_len, VectorRegister *working_vregs, Register blocks,
3070                                        VectorRegister vtmp1, VectorRegister vtmp2, VectorRegister vtmp3) {
3071     __ srli(blocks, input_len, 4);
3072 
3073     const unsigned int BLOCK_SIZE = 16;
3074     const unsigned int MASK_VALUE = 0b1000; // we need {1, 0, 0, 0} mask value here
3075     __ vsetivli(x0, 1, Assembler::e8, Assembler::m1);
3076     __ vmv_v_i(v0, MASK_VALUE);
3077 
3078     __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
3079     // load keys to working_vregs according to round
3080     aes_load_keys(key, working_vregs, round);
3081 
3082     __ vle32_v(vtmp1, counter);
3083     Label L_aes_ctr_loop;
3084     __ bind(L_aes_ctr_loop);
3085       __ vmv_v_v(vtmp2, vtmp1);
3086       // encrypt counter according to round
3087       aes_encrypt(vtmp2, working_vregs, round);
3088       __ vle32_v(vtmp3, in);
3089       __ vxor_vv(vtmp2, vtmp2, vtmp3);
3090       __ vse32_v(vtmp2, out);
3091       __ addi(out, out, BLOCK_SIZE);
3092       __ addi(in, in, BLOCK_SIZE);
3093       __ sub(blocks, blocks, 1);
3094       __ vrev8_v(vtmp1, vtmp1, Assembler::VectorMask::v0_t);
3095       __ vadd_vi(vtmp1, vtmp1, 0x1, Assembler::VectorMask::v0_t);
3096       __ vrev8_v(vtmp1, vtmp1, Assembler::VectorMask::v0_t);
3097       __ bnez(blocks, L_aes_ctr_loop);
3098 
3099     __ vse32_v(vtmp1, counter);
3100   }
3101 
3102   void gcm_ghash_blocks(Register state, Register subkeyH, Register ct, Register input_len, Register blocks,
3103                         VectorRegister vtmp1, VectorRegister vtmp2, VectorRegister vtmp3) {
3104     __ srli(blocks, input_len, 4);
3105 
3106     ghash_loop(state, subkeyH, ct, blocks, vtmp1, vtmp2, vtmp3);
3107 
3108     __ mv(x10, input_len);
3109     __ leave();
3110     __ ret();
3111   }
3112 
3113 
3114   // Vector AES Galois Counter Mode implementation. Parameters:
3115   //
3116   // in = c_rarg0
3117   // input_len = c_rarg1
3118   // ct = c_rarg2 - ciphertext that ghash will read (out for encrypt, in for decrypt)
3119   // out = c_rarg3
3120   // key = c_rarg4
3121   // state = c_rarg5 - GHASH.state
3122   // subkeyHtbl = c_rarg6 - powers of H
3123   // counter = c_rarg7 - 16 bytes of CTR
3124   // return - number of processed bytes
3125   address generate_galoisCounterMode_AESCrypt() {
3126     assert(UseGHASHIntrinsics, "need GHASH instructions (Zvkg extension) and Zvbb support");
3127     assert(UseAESCTRIntrinsics, "need AES instructions (Zvkned extension) and Zbb extension support");
3128 
3129     __ align(CodeEntryAlignment);
3130     StubId stub_id = StubId::stubgen_galoisCounterMode_AESCrypt_id;
3131     StubCodeMark mark(this, stub_id);
3132 
3133     const Register in         = c_rarg0;
3134     const Register input_len  = c_rarg1;
3135     const Register ct         = c_rarg2;
3136     const Register out        = c_rarg3;
3137     const Register key        = c_rarg4;
3138     const Register state      = c_rarg5;
3139     const Register subkeyHtbl = c_rarg6;
3140     const Register counter    = c_rarg7;
3141 
3142     const Register keylen     = x28;
3143     const Register blocks     = x29;
3144 
3145     VectorRegister working_vregs[] = {
3146       v1, v2, v3, v4, v5, v6, v7, v8,
3147       v9, v10, v11, v12, v13, v14, v15
3148     };
3149 
3150     VectorRegister vtmp1      = v16;
3151     VectorRegister vtmp2      = v17;
3152     VectorRegister vtmp3      = v18;
3153 
3154     const address start = __ pc();
3155     __ enter();
3156 
3157     Label L_exit;
3158     // Requires input_len (512) bytes to efficiently use the intrinsic
3159     __ andi(input_len, input_len, -512);
3160     __ beqz(input_len, L_exit);
3161 
3162     Label L_aes128, L_aes192;
3163     // Compute #rounds for AES based on the length of the key array
3164     __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
3165     __ mv(t0, 52); // key length could be only {11, 13, 15} * 4 = {44, 52, 60}
3166     __ bltu(keylen, t0, L_aes128);
3167     __ beq(keylen, t0, L_aes192);
3168     // Else we fallthrough to the biggest case (256-bit key size)
3169 
3170     // Note: the following function performs crypt with key += 15*16
3171     gcm_counterMode_AESCrypt_blocks(15, in, out, key, counter, input_len, working_vregs, blocks, vtmp1, vtmp2, vtmp3);
3172     gcm_ghash_blocks(state, subkeyHtbl, ct, input_len, blocks, vtmp1, vtmp2, vtmp3);
3173 
3174     // Note: the following function performs crypt with key += 13*16
3175     __ bind(L_aes192);
3176     gcm_counterMode_AESCrypt_blocks(13, in, out, key, counter, input_len, working_vregs, blocks, vtmp1, vtmp2, vtmp3);
3177     gcm_ghash_blocks(state, subkeyHtbl, ct, input_len, blocks, vtmp1, vtmp2, vtmp3);
3178 
3179     // Note: the following function performs crypt with key += 11*16
3180     __ bind(L_aes128);
3181     gcm_counterMode_AESCrypt_blocks(11, in, out, key, counter, input_len, working_vregs, blocks, vtmp1, vtmp2, vtmp3);
3182     gcm_ghash_blocks(state, subkeyHtbl, ct, input_len, blocks, vtmp1, vtmp2, vtmp3);
3183 
3184     __ bind(L_exit);
3185     __ mv(x10, input_len);
3186     __ leave();
3187     __ ret();
3188 
3189     return start;
3190   }
3191 
3192   // code for comparing 8 characters of strings with Latin1 and Utf16 encoding
3193   void compare_string_8_x_LU(Register tmpL, Register tmpU,
3194                              Register strL, Register strU, Label& DIFF) {
3195     const Register tmp = x30, tmpLval = x12;
3196 
3197     int base_offset = arrayOopDesc::base_offset_in_bytes(T_BYTE);
3198     assert((base_offset % (UseCompactObjectHeaders ? 4 : 8)) == 0, "Must be");
3199 
3200 #ifdef ASSERT
3201     if (AvoidUnalignedAccesses) {
3202       Label align_ok;
3203       __ andi(t0, strL, 0x7);
3204       __ beqz(t0, align_ok);
3205       __ stop("bad alignment");
3206       __ bind(align_ok);
3207     }
3208 #endif
3209     __ ld(tmpLval, Address(strL));
3210     __ addi(strL, strL, wordSize);
3211 
3212     // compare first 4 characters
3213     __ load_long_misaligned(tmpU, Address(strU), tmp, (base_offset % 8) != 0 ? 4 : 8);
3214     __ addi(strU, strU, wordSize);
3215     __ inflate_lo32(tmpL, tmpLval);
3216     __ xorr(tmp, tmpU, tmpL);
3217     __ bnez(tmp, DIFF);
3218 
3219     // compare second 4 characters
3220     __ load_long_misaligned(tmpU, Address(strU), tmp, (base_offset % 8) != 0 ? 4 : 8);
3221     __ addi(strU, strU, wordSize);
3222     __ inflate_hi32(tmpL, tmpLval);
3223     __ xorr(tmp, tmpU, tmpL);
3224     __ bnez(tmp, DIFF);
3225   }
3226 
3227   // x10  = result
3228   // x11  = str1
3229   // x12  = cnt1
3230   // x13  = str2
3231   // x14  = cnt2
3232   // x28  = tmp1
3233   // x29  = tmp2
3234   // x30  = tmp3
3235   address generate_compare_long_string_different_encoding(StubId stub_id) {
3236     bool isLU;
3237     switch (stub_id) {
3238     case StubId::stubgen_compare_long_string_LU_id:
3239       isLU = true;
3240       break;
3241     case StubId::stubgen_compare_long_string_UL_id:
3242       isLU = false;
3243       break;
3244     default:
3245       ShouldNotReachHere();
3246     };
3247     __ align(CodeEntryAlignment);
3248     StubCodeMark mark(this, stub_id);
3249     address entry = __ pc();
3250     Label SMALL_LOOP, TAIL, LOAD_LAST, DONE, CALCULATE_DIFFERENCE;
3251     const Register result = x10, str1 = x11, str2 = x13, cnt2 = x14,
3252                    tmp1 = x28, tmp2 = x29, tmp3 = x30, tmp4 = x12;
3253 
3254     int base_offset = arrayOopDesc::base_offset_in_bytes(T_BYTE);
3255     assert((base_offset % (UseCompactObjectHeaders ? 4 : 8)) == 0, "Must be");
3256 
3257     Register strU = isLU ? str2 : str1,
3258              strL = isLU ? str1 : str2,
3259              tmpU = isLU ? tmp2 : tmp1, // where to keep U for comparison
3260              tmpL = isLU ? tmp1 : tmp2; // where to keep L for comparison
3261 
3262     if (AvoidUnalignedAccesses && (base_offset % 8) != 0) {
3263       // Load 4 bytes from strL to make sure main loop is 8-byte aligned
3264       // cnt2 is >= 68 here, no need to check it for >= 0
3265       __ lwu(tmpL, Address(strL));
3266       __ addi(strL, strL, wordSize / 2);
3267       __ load_long_misaligned(tmpU, Address(strU), tmp4, (base_offset % 8) != 0 ? 4 : 8);
3268       __ addi(strU, strU, wordSize);
3269       __ inflate_lo32(tmp3, tmpL);
3270       __ mv(tmpL, tmp3);
3271       __ xorr(tmp3, tmpU, tmpL);
3272       __ bnez(tmp3, CALCULATE_DIFFERENCE);
3273       __ subi(cnt2, cnt2, wordSize / 2);
3274     }
3275 
3276     // we are now 8-bytes aligned on strL when AvoidUnalignedAccesses is true
3277     __ subi(cnt2, cnt2, wordSize * 2);
3278     __ bltz(cnt2, TAIL);
3279     __ bind(SMALL_LOOP); // smaller loop
3280       __ subi(cnt2, cnt2, wordSize * 2);
3281       compare_string_8_x_LU(tmpL, tmpU, strL, strU, CALCULATE_DIFFERENCE);
3282       compare_string_8_x_LU(tmpL, tmpU, strL, strU, CALCULATE_DIFFERENCE);
3283       __ bgez(cnt2, SMALL_LOOP);
3284       __ addi(t0, cnt2, wordSize * 2);
3285       __ beqz(t0, DONE);
3286     __ bind(TAIL);  // 1..15 characters left
3287       // Aligned access. Load bytes in portions - 4, 2, 1.
3288 
3289       __ addi(t0, cnt2, wordSize);
3290       __ addi(cnt2, cnt2, wordSize * 2); // amount of characters left to process
3291       __ bltz(t0, LOAD_LAST);
3292       // remaining characters are greater than or equals to 8, we can do one compare_string_8_x_LU
3293       compare_string_8_x_LU(tmpL, tmpU, strL, strU, CALCULATE_DIFFERENCE);
3294       __ subi(cnt2, cnt2, wordSize);
3295       __ beqz(cnt2, DONE);  // no character left
3296       __ bind(LOAD_LAST);   // cnt2 = 1..7 characters left
3297 
3298       __ subi(cnt2, cnt2, wordSize); // cnt2 is now an offset in strL which points to last 8 bytes
3299       __ slli(t0, cnt2, 1);     // t0 is now an offset in strU which points to last 16 bytes
3300       __ add(strL, strL, cnt2); // Address of last 8 bytes in Latin1 string
3301       __ add(strU, strU, t0);   // Address of last 16 bytes in UTF-16 string
3302       __ load_int_misaligned(tmpL, Address(strL), t0, false);
3303       __ load_long_misaligned(tmpU, Address(strU), t0, 2);
3304       __ inflate_lo32(tmp3, tmpL);
3305       __ mv(tmpL, tmp3);
3306       __ xorr(tmp3, tmpU, tmpL);
3307       __ bnez(tmp3, CALCULATE_DIFFERENCE);
3308 
3309       __ addi(strL, strL, wordSize / 2); // Address of last 4 bytes in Latin1 string
3310       __ addi(strU, strU, wordSize);   // Address of last 8 bytes in UTF-16 string
3311       __ load_int_misaligned(tmpL, Address(strL), t0, false);
3312       __ load_long_misaligned(tmpU, Address(strU), t0, 2);
3313       __ inflate_lo32(tmp3, tmpL);
3314       __ mv(tmpL, tmp3);
3315       __ xorr(tmp3, tmpU, tmpL);
3316       __ bnez(tmp3, CALCULATE_DIFFERENCE);
3317       __ j(DONE); // no character left
3318 
3319       // Find the first different characters in the longwords and
3320       // compute their difference.
3321     __ bind(CALCULATE_DIFFERENCE);
3322       // count bits of trailing zero chars
3323       __ ctzc_bits(tmp4, tmp3);
3324       __ srl(tmp1, tmp1, tmp4);
3325       __ srl(tmp2, tmp2, tmp4);
3326       __ zext(tmp1, tmp1, 16);
3327       __ zext(tmp2, tmp2, 16);
3328       __ sub(result, tmp1, tmp2);
3329     __ bind(DONE);
3330       __ ret();
3331     return entry;
3332   }
3333 
3334   address generate_method_entry_barrier() {
3335     __ align(CodeEntryAlignment);
3336     StubId stub_id = StubId::stubgen_method_entry_barrier_id;
3337     StubCodeMark mark(this, stub_id);
3338 
3339     Label deoptimize_label;
3340 
3341     address start = __ pc();
3342 
3343     BarrierSetAssembler* bs_asm = BarrierSet::barrier_set()->barrier_set_assembler();
3344 
3345     if (bs_asm->nmethod_patching_type() == NMethodPatchingType::conc_instruction_and_data_patch) {
3346       BarrierSetNMethod* bs_nm = BarrierSet::barrier_set()->barrier_set_nmethod();
3347       Address thread_epoch_addr(xthread, in_bytes(bs_nm->thread_disarmed_guard_value_offset()) + 4);
3348       __ la(t1, ExternalAddress(bs_asm->patching_epoch_addr()));
3349       __ lwu(t1, t1);
3350       __ sw(t1, thread_epoch_addr);
3351       // There are two ways this can work:
3352       // - The writer did system icache shootdown after the instruction stream update.
3353       //   Hence do nothing.
3354       // - The writer trust us to make sure our icache is in sync before entering.
3355       //   Hence use cmodx fence (fence.i, may change).
3356       if (UseCtxFencei) {
3357         __ cmodx_fence();
3358       }
3359       __ membar(__ LoadLoad);
3360     }
3361 
3362     __ set_last_Java_frame(sp, fp, ra);
3363 
3364     __ enter();
3365     __ addi(t1, sp, wordSize);
3366 
3367     __ subi(sp, sp, 4 * wordSize);
3368 
3369     __ push_call_clobbered_registers();
3370 
3371     __ mv(c_rarg0, t1);
3372     __ call_VM_leaf(CAST_FROM_FN_PTR(address, BarrierSetNMethod::nmethod_stub_entry_barrier), 1);
3373 
3374     __ reset_last_Java_frame(true);
3375 
3376     __ mv(t0, x10);
3377 
3378     __ pop_call_clobbered_registers();
3379 
3380     __ bnez(t0, deoptimize_label);
3381 
3382     __ leave();
3383     __ ret();
3384 
3385     __ BIND(deoptimize_label);
3386 
3387     __ ld(t0, Address(sp, 0));
3388     __ ld(fp, Address(sp, wordSize));
3389     __ ld(ra, Address(sp, wordSize * 2));
3390     __ ld(t1, Address(sp, wordSize * 3));
3391 
3392     __ mv(sp, t0);
3393     __ jr(t1);
3394 
3395     return start;
3396   }
3397 
3398   // x10  = result
3399   // x11  = str1
3400   // x12  = cnt1
3401   // x13  = str2
3402   // x14  = cnt2
3403   // x28  = tmp1
3404   // x29  = tmp2
3405   // x30  = tmp3
3406   // x31  = tmp4
3407   address generate_compare_long_string_same_encoding(StubId stub_id) {
3408     bool isLL;
3409     switch (stub_id) {
3410     case StubId::stubgen_compare_long_string_LL_id:
3411       isLL = true;
3412       break;
3413     case StubId::stubgen_compare_long_string_UU_id:
3414       isLL = false;
3415       break;
3416     default:
3417       ShouldNotReachHere();
3418     };
3419     __ align(CodeEntryAlignment);
3420     StubCodeMark mark(this, stub_id);
3421     address entry = __ pc();
3422     Label SMALL_LOOP, CHECK_LAST, DIFF2, TAIL,
3423           LENGTH_DIFF, DIFF, LAST_CHECK_AND_LENGTH_DIFF;
3424     const Register result = x10, str1 = x11, cnt1 = x12, str2 = x13, cnt2 = x14,
3425                    tmp1 = x28, tmp2 = x29, tmp3 = x30, tmp4 = x7, tmp5 = x31;
3426     RegSet spilled_regs = RegSet::of(tmp4, tmp5);
3427 
3428     // cnt1/cnt2 contains amount of characters to compare. cnt1 can be re-used
3429     // update cnt2 counter with already loaded 8 bytes
3430     __ subi(cnt2, cnt2, wordSize / (isLL ? 1 : 2));
3431     // update pointers, because of previous read
3432     __ addi(str1, str1, wordSize);
3433     __ addi(str2, str2, wordSize);
3434     // less than 16 bytes left?
3435     __ subi(cnt2, cnt2, isLL ? 16 : 8);
3436     __ push_reg(spilled_regs, sp);
3437     __ bltz(cnt2, TAIL);
3438     __ bind(SMALL_LOOP);
3439       // compare 16 bytes of strings with same encoding
3440       __ ld(tmp5, Address(str1));
3441       __ addi(str1, str1, 8);
3442       __ xorr(tmp4, tmp1, tmp2);
3443       __ ld(cnt1, Address(str2));
3444       __ addi(str2, str2, 8);
3445       __ bnez(tmp4, DIFF);
3446       __ ld(tmp1, Address(str1));
3447       __ addi(str1, str1, 8);
3448       __ xorr(tmp4, tmp5, cnt1);
3449       __ ld(tmp2, Address(str2));
3450       __ addi(str2, str2, 8);
3451       __ bnez(tmp4, DIFF2);
3452 
3453       __ subi(cnt2, cnt2, isLL ? 16 : 8);
3454       __ bgez(cnt2, SMALL_LOOP);
3455     __ bind(TAIL);
3456       __ addi(cnt2, cnt2, isLL ? 16 : 8);
3457       __ beqz(cnt2, LAST_CHECK_AND_LENGTH_DIFF);
3458       __ subi(cnt2, cnt2, isLL ? 8 : 4);
3459       __ blez(cnt2, CHECK_LAST);
3460       __ xorr(tmp4, tmp1, tmp2);
3461       __ bnez(tmp4, DIFF);
3462       __ ld(tmp1, Address(str1));
3463       __ addi(str1, str1, 8);
3464       __ ld(tmp2, Address(str2));
3465       __ addi(str2, str2, 8);
3466       __ subi(cnt2, cnt2, isLL ? 8 : 4);
3467     __ bind(CHECK_LAST);
3468       if (!isLL) {
3469         __ add(cnt2, cnt2, cnt2); // now in bytes
3470       }
3471       __ xorr(tmp4, tmp1, tmp2);
3472       __ bnez(tmp4, DIFF);
3473       __ add(str1, str1, cnt2);
3474       __ load_long_misaligned(tmp5, Address(str1), tmp3, isLL ? 1 : 2);
3475       __ add(str2, str2, cnt2);
3476       __ load_long_misaligned(cnt1, Address(str2), tmp3, isLL ? 1 : 2);
3477       __ xorr(tmp4, tmp5, cnt1);
3478       __ beqz(tmp4, LENGTH_DIFF);
3479       // Find the first different characters in the longwords and
3480       // compute their difference.
3481     __ bind(DIFF2);
3482       // count bits of trailing zero chars
3483       __ ctzc_bits(tmp3, tmp4, isLL);
3484       __ srl(tmp5, tmp5, tmp3);
3485       __ srl(cnt1, cnt1, tmp3);
3486       if (isLL) {
3487         __ zext(tmp5, tmp5, 8);
3488         __ zext(cnt1, cnt1, 8);
3489       } else {
3490         __ zext(tmp5, tmp5, 16);
3491         __ zext(cnt1, cnt1, 16);
3492       }
3493       __ sub(result, tmp5, cnt1);
3494       __ j(LENGTH_DIFF);
3495     __ bind(DIFF);
3496       // count bits of trailing zero chars
3497       __ ctzc_bits(tmp3, tmp4, isLL);
3498       __ srl(tmp1, tmp1, tmp3);
3499       __ srl(tmp2, tmp2, tmp3);
3500       if (isLL) {
3501         __ zext(tmp1, tmp1, 8);
3502         __ zext(tmp2, tmp2, 8);
3503       } else {
3504         __ zext(tmp1, tmp1, 16);
3505         __ zext(tmp2, tmp2, 16);
3506       }
3507       __ sub(result, tmp1, tmp2);
3508       __ j(LENGTH_DIFF);
3509     __ bind(LAST_CHECK_AND_LENGTH_DIFF);
3510       __ xorr(tmp4, tmp1, tmp2);
3511       __ bnez(tmp4, DIFF);
3512     __ bind(LENGTH_DIFF);
3513       __ pop_reg(spilled_regs, sp);
3514       __ ret();
3515     return entry;
3516   }
3517 
3518   void generate_compare_long_strings() {
3519     StubRoutines::riscv::_compare_long_string_LL = generate_compare_long_string_same_encoding(StubId::stubgen_compare_long_string_LL_id);
3520     StubRoutines::riscv::_compare_long_string_UU = generate_compare_long_string_same_encoding(StubId::stubgen_compare_long_string_UU_id);
3521     StubRoutines::riscv::_compare_long_string_LU = generate_compare_long_string_different_encoding(StubId::stubgen_compare_long_string_LU_id);
3522     StubRoutines::riscv::_compare_long_string_UL = generate_compare_long_string_different_encoding(StubId::stubgen_compare_long_string_UL_id);
3523   }
3524 
3525   // x10 result
3526   // x11 src
3527   // x12 src count
3528   // x13 pattern
3529   // x14 pattern count
3530   address generate_string_indexof_linear(StubId stub_id)
3531   {
3532     bool needle_isL;
3533     bool haystack_isL;
3534     switch (stub_id) {
3535     case StubId::stubgen_string_indexof_linear_ll_id:
3536       needle_isL = true;
3537       haystack_isL = true;
3538       break;
3539     case StubId::stubgen_string_indexof_linear_ul_id:
3540       needle_isL = true;
3541       haystack_isL = false;
3542       break;
3543     case StubId::stubgen_string_indexof_linear_uu_id:
3544       needle_isL = false;
3545       haystack_isL = false;
3546       break;
3547     default:
3548       ShouldNotReachHere();
3549     };
3550 
3551     __ align(CodeEntryAlignment);
3552     StubCodeMark mark(this, stub_id);
3553     address entry = __ pc();
3554 
3555     int needle_chr_size = needle_isL ? 1 : 2;
3556     int haystack_chr_size = haystack_isL ? 1 : 2;
3557     int needle_chr_shift = needle_isL ? 0 : 1;
3558     int haystack_chr_shift = haystack_isL ? 0 : 1;
3559     bool isL = needle_isL && haystack_isL;
3560     // parameters
3561     Register result = x10, haystack = x11, haystack_len = x12, needle = x13, needle_len = x14;
3562     // temporary registers
3563     Register mask1 = x20, match_mask = x21, first = x22, trailing_zeros = x23, mask2 = x24, tmp = x25;
3564     // redefinitions
3565     Register ch1 = x28, ch2 = x29;
3566     RegSet spilled_regs = RegSet::range(x20, x25) + RegSet::range(x28, x29);
3567 
3568     __ push_reg(spilled_regs, sp);
3569 
3570     Label L_LOOP, L_LOOP_PROCEED, L_SMALL, L_HAS_ZERO,
3571           L_HAS_ZERO_LOOP, L_CMP_LOOP, L_CMP_LOOP_NOMATCH, L_SMALL_PROCEED,
3572           L_SMALL_HAS_ZERO_LOOP, L_SMALL_CMP_LOOP_NOMATCH, L_SMALL_CMP_LOOP,
3573           L_POST_LOOP, L_CMP_LOOP_LAST_CMP, L_HAS_ZERO_LOOP_NOMATCH,
3574           L_SMALL_CMP_LOOP_LAST_CMP, L_SMALL_CMP_LOOP_LAST_CMP2,
3575           L_CMP_LOOP_LAST_CMP2, DONE, NOMATCH;
3576 
3577     __ ld(ch1, Address(needle));
3578     __ ld(ch2, Address(haystack));
3579     // src.length - pattern.length
3580     __ sub(haystack_len, haystack_len, needle_len);
3581 
3582     // first is needle[0]
3583     __ zext(first, ch1, needle_isL ? 8 : 16);
3584 
3585     uint64_t mask0101 = UCONST64(0x0101010101010101);
3586     uint64_t mask0001 = UCONST64(0x0001000100010001);
3587     __ mv(mask1, haystack_isL ? mask0101 : mask0001);
3588     __ mul(first, first, mask1);
3589     uint64_t mask7f7f = UCONST64(0x7f7f7f7f7f7f7f7f);
3590     uint64_t mask7fff = UCONST64(0x7fff7fff7fff7fff);
3591     __ mv(mask2, haystack_isL ? mask7f7f : mask7fff);
3592     if (needle_isL != haystack_isL) {
3593       __ mv(tmp, ch1);
3594     }
3595     __ subi(haystack_len, haystack_len, wordSize / haystack_chr_size - 1);
3596     __ blez(haystack_len, L_SMALL);
3597 
3598     if (needle_isL != haystack_isL) {
3599       __ inflate_lo32(ch1, tmp, match_mask, trailing_zeros);
3600     }
3601     // xorr, sub, orr, notr, andr
3602     // compare and set match_mask[i] with 0x80/0x8000 (Latin1/UTF16) if ch2[i] == first[i]
3603     // eg:
3604     // first:        aa aa aa aa aa aa aa aa
3605     // ch2:          aa aa li nx jd ka aa aa
3606     // match_mask:   80 80 00 00 00 00 80 80
3607     __ compute_match_mask(ch2, first, match_mask, mask1, mask2);
3608 
3609     // search first char of needle, if success, goto L_HAS_ZERO;
3610     __ bnez(match_mask, L_HAS_ZERO);
3611     __ subi(haystack_len, haystack_len, wordSize / haystack_chr_size);
3612     __ addi(result, result, wordSize / haystack_chr_size);
3613     __ addi(haystack, haystack, wordSize);
3614     __ bltz(haystack_len, L_POST_LOOP);
3615 
3616     __ bind(L_LOOP);
3617     __ ld(ch2, Address(haystack));
3618     __ compute_match_mask(ch2, first, match_mask, mask1, mask2);
3619     __ bnez(match_mask, L_HAS_ZERO);
3620 
3621     __ bind(L_LOOP_PROCEED);
3622     __ subi(haystack_len, haystack_len, wordSize / haystack_chr_size);
3623     __ addi(haystack, haystack, wordSize);
3624     __ addi(result, result, wordSize / haystack_chr_size);
3625     __ bgez(haystack_len, L_LOOP);
3626 
3627     __ bind(L_POST_LOOP);
3628     __ mv(ch2, -wordSize / haystack_chr_size);
3629     __ ble(haystack_len, ch2, NOMATCH); // no extra characters to check
3630     __ ld(ch2, Address(haystack));
3631     __ slli(haystack_len, haystack_len, LogBitsPerByte + haystack_chr_shift);
3632     __ neg(haystack_len, haystack_len);
3633     __ xorr(ch2, first, ch2);
3634     __ sub(match_mask, ch2, mask1);
3635     __ orr(ch2, ch2, mask2);
3636     __ mv(trailing_zeros, -1); // all bits set
3637     __ j(L_SMALL_PROCEED);
3638 
3639     __ align(OptoLoopAlignment);
3640     __ bind(L_SMALL);
3641     __ slli(haystack_len, haystack_len, LogBitsPerByte + haystack_chr_shift);
3642     __ neg(haystack_len, haystack_len);
3643     if (needle_isL != haystack_isL) {
3644       __ inflate_lo32(ch1, tmp, match_mask, trailing_zeros);
3645     }
3646     __ xorr(ch2, first, ch2);
3647     __ sub(match_mask, ch2, mask1);
3648     __ orr(ch2, ch2, mask2);
3649     __ mv(trailing_zeros, -1); // all bits set
3650 
3651     __ bind(L_SMALL_PROCEED);
3652     __ srl(trailing_zeros, trailing_zeros, haystack_len); // mask. zeroes on useless bits.
3653     __ notr(ch2, ch2);
3654     __ andr(match_mask, match_mask, ch2);
3655     __ andr(match_mask, match_mask, trailing_zeros); // clear useless bits and check
3656     __ beqz(match_mask, NOMATCH);
3657 
3658     __ bind(L_SMALL_HAS_ZERO_LOOP);
3659     // count bits of trailing zero chars
3660     __ ctzc_bits(trailing_zeros, match_mask, haystack_isL, ch2, tmp);
3661     __ addi(trailing_zeros, trailing_zeros, haystack_isL ? 7 : 15);
3662     __ mv(ch2, wordSize / haystack_chr_size);
3663     __ ble(needle_len, ch2, L_SMALL_CMP_LOOP_LAST_CMP2);
3664     __ compute_index(haystack, trailing_zeros, match_mask, result, ch2, tmp, haystack_isL);
3665     __ mv(trailing_zeros, wordSize / haystack_chr_size);
3666     __ bne(ch1, ch2, L_SMALL_CMP_LOOP_NOMATCH);
3667 
3668     __ bind(L_SMALL_CMP_LOOP);
3669     __ shadd(first, trailing_zeros, needle, first, needle_chr_shift);
3670     __ shadd(ch2, trailing_zeros, haystack, ch2, haystack_chr_shift);
3671     needle_isL ? __ lbu(first, Address(first)) : __ lhu(first, Address(first));
3672     haystack_isL ? __ lbu(ch2, Address(ch2)) : __ lhu(ch2, Address(ch2));
3673     __ addi(trailing_zeros, trailing_zeros, 1);
3674     __ bge(trailing_zeros, needle_len, L_SMALL_CMP_LOOP_LAST_CMP);
3675     __ beq(first, ch2, L_SMALL_CMP_LOOP);
3676 
3677     __ bind(L_SMALL_CMP_LOOP_NOMATCH);
3678     __ beqz(match_mask, NOMATCH);
3679     // count bits of trailing zero chars
3680     __ ctzc_bits(trailing_zeros, match_mask, haystack_isL, tmp, ch2);
3681     __ addi(trailing_zeros, trailing_zeros, haystack_isL ? 7 : 15);
3682     __ addi(result, result, 1);
3683     __ addi(haystack, haystack, haystack_chr_size);
3684     __ j(L_SMALL_HAS_ZERO_LOOP);
3685 
3686     __ align(OptoLoopAlignment);
3687     __ bind(L_SMALL_CMP_LOOP_LAST_CMP);
3688     __ bne(first, ch2, L_SMALL_CMP_LOOP_NOMATCH);
3689     __ j(DONE);
3690 
3691     __ align(OptoLoopAlignment);
3692     __ bind(L_SMALL_CMP_LOOP_LAST_CMP2);
3693     __ compute_index(haystack, trailing_zeros, match_mask, result, ch2, tmp, haystack_isL);
3694     __ bne(ch1, ch2, L_SMALL_CMP_LOOP_NOMATCH);
3695     __ j(DONE);
3696 
3697     __ align(OptoLoopAlignment);
3698     __ bind(L_HAS_ZERO);
3699     // count bits of trailing zero chars
3700     __ ctzc_bits(trailing_zeros, match_mask, haystack_isL, tmp, ch2);
3701     __ addi(trailing_zeros, trailing_zeros, haystack_isL ? 7 : 15);
3702     __ slli(needle_len, needle_len, BitsPerByte * wordSize / 2);
3703     __ orr(haystack_len, haystack_len, needle_len); // restore needle_len(32bits)
3704     __ subi(result, result, 1); // array index from 0, so result -= 1
3705 
3706     __ bind(L_HAS_ZERO_LOOP);
3707     __ mv(needle_len, wordSize / haystack_chr_size);
3708     __ srli(ch2, haystack_len, BitsPerByte * wordSize / 2);
3709     __ bge(needle_len, ch2, L_CMP_LOOP_LAST_CMP2);
3710     // load next 8 bytes from haystack, and increase result index
3711     __ compute_index(haystack, trailing_zeros, match_mask, result, ch2, tmp, haystack_isL);
3712     __ addi(result, result, 1);
3713     __ mv(trailing_zeros, wordSize / haystack_chr_size);
3714     __ bne(ch1, ch2, L_CMP_LOOP_NOMATCH);
3715 
3716     // compare one char
3717     __ bind(L_CMP_LOOP);
3718     __ shadd(needle_len, trailing_zeros, needle, needle_len, needle_chr_shift);
3719     needle_isL ? __ lbu(needle_len, Address(needle_len)) : __ lhu(needle_len, Address(needle_len));
3720     __ shadd(ch2, trailing_zeros, haystack, ch2, haystack_chr_shift);
3721     haystack_isL ? __ lbu(ch2, Address(ch2)) : __ lhu(ch2, Address(ch2));
3722     __ addi(trailing_zeros, trailing_zeros, 1); // next char index
3723     __ srli(tmp, haystack_len, BitsPerByte * wordSize / 2);
3724     __ bge(trailing_zeros, tmp, L_CMP_LOOP_LAST_CMP);
3725     __ beq(needle_len, ch2, L_CMP_LOOP);
3726 
3727     __ bind(L_CMP_LOOP_NOMATCH);
3728     __ beqz(match_mask, L_HAS_ZERO_LOOP_NOMATCH);
3729     // count bits of trailing zero chars
3730     __ ctzc_bits(trailing_zeros, match_mask, haystack_isL, needle_len, ch2);
3731     __ addi(trailing_zeros, trailing_zeros, haystack_isL ? 7 : 15);
3732     __ addi(haystack, haystack, haystack_chr_size);
3733     __ j(L_HAS_ZERO_LOOP);
3734 
3735     __ align(OptoLoopAlignment);
3736     __ bind(L_CMP_LOOP_LAST_CMP);
3737     __ bne(needle_len, ch2, L_CMP_LOOP_NOMATCH);
3738     __ j(DONE);
3739 
3740     __ align(OptoLoopAlignment);
3741     __ bind(L_CMP_LOOP_LAST_CMP2);
3742     __ compute_index(haystack, trailing_zeros, match_mask, result, ch2, tmp, haystack_isL);
3743     __ addi(result, result, 1);
3744     __ bne(ch1, ch2, L_CMP_LOOP_NOMATCH);
3745     __ j(DONE);
3746 
3747     __ align(OptoLoopAlignment);
3748     __ bind(L_HAS_ZERO_LOOP_NOMATCH);
3749     // 1) Restore "result" index. Index was wordSize/str2_chr_size * N until
3750     // L_HAS_ZERO block. Byte octet was analyzed in L_HAS_ZERO_LOOP,
3751     // so, result was increased at max by wordSize/str2_chr_size - 1, so,
3752     // respective high bit wasn't changed. L_LOOP_PROCEED will increase
3753     // result by analyzed characters value, so, we can just reset lower bits
3754     // in result here. Clear 2 lower bits for UU/UL and 3 bits for LL
3755     // 2) restore needle_len and haystack_len values from "compressed" haystack_len
3756     // 3) advance haystack value to represent next haystack octet. result & 7/3 is
3757     // index of last analyzed substring inside current octet. So, haystack in at
3758     // respective start address. We need to advance it to next octet
3759     __ andi(match_mask, result, wordSize / haystack_chr_size - 1);
3760     __ srli(needle_len, haystack_len, BitsPerByte * wordSize / 2);
3761     __ andi(result, result, haystack_isL ? -8 : -4);
3762     __ slli(tmp, match_mask, haystack_chr_shift);
3763     __ sub(haystack, haystack, tmp);
3764     __ sext(haystack_len, haystack_len, 32);
3765     __ j(L_LOOP_PROCEED);
3766 
3767     __ align(OptoLoopAlignment);
3768     __ bind(NOMATCH);
3769     __ mv(result, -1);
3770 
3771     __ bind(DONE);
3772     __ pop_reg(spilled_regs, sp);
3773     __ ret();
3774     return entry;
3775   }
3776 
3777   void generate_string_indexof_stubs()
3778   {
3779     StubRoutines::riscv::_string_indexof_linear_ll = generate_string_indexof_linear(StubId::stubgen_string_indexof_linear_ll_id);
3780     StubRoutines::riscv::_string_indexof_linear_uu = generate_string_indexof_linear(StubId::stubgen_string_indexof_linear_uu_id);
3781     StubRoutines::riscv::_string_indexof_linear_ul = generate_string_indexof_linear(StubId::stubgen_string_indexof_linear_ul_id);
3782   }
3783 
3784 #ifdef COMPILER2
3785   void generate_lookup_secondary_supers_table_stub() {
3786     StubId stub_id = StubId::stubgen_lookup_secondary_supers_table_id;
3787     StubCodeMark mark(this, stub_id);
3788 
3789     const Register
3790       r_super_klass  = x10,
3791       r_array_base   = x11,
3792       r_array_length = x12,
3793       r_array_index  = x13,
3794       r_sub_klass    = x14,
3795       result         = x15,
3796       r_bitmap       = x16;
3797 
3798     for (int slot = 0; slot < Klass::SECONDARY_SUPERS_TABLE_SIZE; slot++) {
3799       StubRoutines::_lookup_secondary_supers_table_stubs[slot] = __ pc();
3800       Label L_success;
3801       __ enter();
3802       __ lookup_secondary_supers_table_const(r_sub_klass, r_super_klass, result,
3803                                              r_array_base, r_array_length, r_array_index,
3804                                              r_bitmap, slot, /*stub_is_near*/true);
3805       __ leave();
3806       __ ret();
3807     }
3808   }
3809 
3810   // Slow path implementation for UseSecondarySupersTable.
3811   address generate_lookup_secondary_supers_table_slow_path_stub() {
3812     StubId stub_id = StubId::stubgen_lookup_secondary_supers_table_slow_path_id;
3813     StubCodeMark mark(this, stub_id);
3814 
3815     address start = __ pc();
3816     const Register
3817       r_super_klass  = x10,        // argument
3818       r_array_base   = x11,        // argument
3819       temp1          = x12,        // tmp
3820       r_array_index  = x13,        // argument
3821       result         = x15,        // argument
3822       r_bitmap       = x16;        // argument
3823 
3824 
3825     __ lookup_secondary_supers_table_slow_path(r_super_klass, r_array_base, r_array_index, r_bitmap, result, temp1);
3826     __ ret();
3827 
3828     return start;
3829   }
3830 
3831   address generate_mulAdd()
3832   {
3833     __ align(CodeEntryAlignment);
3834     StubId stub_id = StubId::stubgen_mulAdd_id;
3835     StubCodeMark mark(this, stub_id);
3836 
3837     address entry = __ pc();
3838 
3839     const Register out     = x10;
3840     const Register in      = x11;
3841     const Register offset  = x12;
3842     const Register len     = x13;
3843     const Register k       = x14;
3844     const Register tmp     = x28;
3845 
3846     BLOCK_COMMENT("Entry:");
3847     __ enter();
3848     __ mul_add(out, in, offset, len, k, tmp);
3849     __ leave();
3850     __ ret();
3851 
3852     return entry;
3853   }
3854 
3855   /**
3856    *  Arguments:
3857    *
3858    *  Input:
3859    *    c_rarg0   - x address
3860    *    c_rarg1   - x length
3861    *    c_rarg2   - y address
3862    *    c_rarg3   - y length
3863    *    c_rarg4   - z address
3864    */
3865   address generate_multiplyToLen()
3866   {
3867     __ align(CodeEntryAlignment);
3868     StubId stub_id = StubId::stubgen_multiplyToLen_id;
3869     StubCodeMark mark(this, stub_id);
3870     address entry = __ pc();
3871 
3872     const Register x     = x10;
3873     const Register xlen  = x11;
3874     const Register y     = x12;
3875     const Register ylen  = x13;
3876     const Register z     = x14;
3877 
3878     const Register tmp0  = x15;
3879     const Register tmp1  = x16;
3880     const Register tmp2  = x17;
3881     const Register tmp3  = x7;
3882     const Register tmp4  = x28;
3883     const Register tmp5  = x29;
3884     const Register tmp6  = x30;
3885     const Register tmp7  = x31;
3886 
3887     BLOCK_COMMENT("Entry:");
3888     __ enter(); // required for proper stackwalking of RuntimeStub frame
3889     __ multiply_to_len(x, xlen, y, ylen, z, tmp0, tmp1, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7);
3890     __ leave(); // required for proper stackwalking of RuntimeStub frame
3891     __ ret();
3892 
3893     return entry;
3894   }
3895 
3896   address generate_squareToLen()
3897   {
3898     __ align(CodeEntryAlignment);
3899     StubId stub_id = StubId::stubgen_squareToLen_id;
3900     StubCodeMark mark(this, stub_id);
3901     address entry = __ pc();
3902 
3903     const Register x     = x10;
3904     const Register xlen  = x11;
3905     const Register z     = x12;
3906     const Register y     = x14; // == x
3907     const Register ylen  = x15; // == xlen
3908 
3909     const Register tmp0  = x13; // zlen, unused
3910     const Register tmp1  = x16;
3911     const Register tmp2  = x17;
3912     const Register tmp3  = x7;
3913     const Register tmp4  = x28;
3914     const Register tmp5  = x29;
3915     const Register tmp6  = x30;
3916     const Register tmp7  = x31;
3917 
3918     BLOCK_COMMENT("Entry:");
3919     __ enter();
3920     __ mv(y, x);
3921     __ mv(ylen, xlen);
3922     __ multiply_to_len(x, xlen, y, ylen, z, tmp0, tmp1, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7);
3923     __ leave();
3924     __ ret();
3925 
3926     return entry;
3927   }
3928 
3929   // Arguments:
3930   //
3931   // Input:
3932   //   c_rarg0   - newArr address
3933   //   c_rarg1   - oldArr address
3934   //   c_rarg2   - newIdx
3935   //   c_rarg3   - shiftCount
3936   //   c_rarg4   - numIter
3937   //
3938   address generate_bigIntegerLeftShift() {
3939     __ align(CodeEntryAlignment);
3940     StubId stub_id = StubId::stubgen_bigIntegerLeftShiftWorker_id;
3941     StubCodeMark mark(this, stub_id);
3942     address entry = __ pc();
3943 
3944     Label loop, exit;
3945 
3946     Register newArr        = c_rarg0;
3947     Register oldArr        = c_rarg1;
3948     Register newIdx        = c_rarg2;
3949     Register shiftCount    = c_rarg3;
3950     Register numIter       = c_rarg4;
3951 
3952     Register shiftRevCount = c_rarg5;
3953     Register oldArrNext    = t1;
3954 
3955     __ beqz(numIter, exit);
3956     __ shadd(newArr, newIdx, newArr, t0, 2);
3957 
3958     __ mv(shiftRevCount, 32);
3959     __ sub(shiftRevCount, shiftRevCount, shiftCount);
3960 
3961     __ bind(loop);
3962     __ addi(oldArrNext, oldArr, 4);
3963     __ vsetvli(t0, numIter, Assembler::e32, Assembler::m4);
3964     __ vle32_v(v0, oldArr);
3965     __ vle32_v(v4, oldArrNext);
3966     __ vsll_vx(v0, v0, shiftCount);
3967     __ vsrl_vx(v4, v4, shiftRevCount);
3968     __ vor_vv(v0, v0, v4);
3969     __ vse32_v(v0, newArr);
3970     __ sub(numIter, numIter, t0);
3971     __ shadd(oldArr, t0, oldArr, t1, 2);
3972     __ shadd(newArr, t0, newArr, t1, 2);
3973     __ bnez(numIter, loop);
3974 
3975     __ bind(exit);
3976     __ ret();
3977 
3978     return entry;
3979   }
3980 
3981   // Arguments:
3982   //
3983   // Input:
3984   //   c_rarg0   - newArr address
3985   //   c_rarg1   - oldArr address
3986   //   c_rarg2   - newIdx
3987   //   c_rarg3   - shiftCount
3988   //   c_rarg4   - numIter
3989   //
3990   address generate_bigIntegerRightShift() {
3991     __ align(CodeEntryAlignment);
3992     StubId stub_id = StubId::stubgen_bigIntegerRightShiftWorker_id;
3993     StubCodeMark mark(this, stub_id);
3994     address entry = __ pc();
3995 
3996     Label loop, exit;
3997 
3998     Register newArr        = c_rarg0;
3999     Register oldArr        = c_rarg1;
4000     Register newIdx        = c_rarg2;
4001     Register shiftCount    = c_rarg3;
4002     Register numIter       = c_rarg4;
4003     Register idx           = numIter;
4004 
4005     Register shiftRevCount = c_rarg5;
4006     Register oldArrNext    = c_rarg6;
4007     Register newArrCur     = t0;
4008     Register oldArrCur     = t1;
4009 
4010     __ beqz(idx, exit);
4011     __ shadd(newArr, newIdx, newArr, t0, 2);
4012 
4013     __ mv(shiftRevCount, 32);
4014     __ sub(shiftRevCount, shiftRevCount, shiftCount);
4015 
4016     __ bind(loop);
4017     __ vsetvli(t0, idx, Assembler::e32, Assembler::m4);
4018     __ sub(idx, idx, t0);
4019     __ shadd(oldArrNext, idx, oldArr, t1, 2);
4020     __ shadd(newArrCur, idx, newArr, t1, 2);
4021     __ addi(oldArrCur, oldArrNext, 4);
4022     __ vle32_v(v0, oldArrCur);
4023     __ vle32_v(v4, oldArrNext);
4024     __ vsrl_vx(v0, v0, shiftCount);
4025     __ vsll_vx(v4, v4, shiftRevCount);
4026     __ vor_vv(v0, v0, v4);
4027     __ vse32_v(v0, newArrCur);
4028     __ bnez(idx, loop);
4029 
4030     __ bind(exit);
4031     __ ret();
4032 
4033     return entry;
4034   }
4035 #endif
4036 
4037 #ifdef COMPILER2
4038   class MontgomeryMultiplyGenerator : public MacroAssembler {
4039 
4040     Register Pa_base, Pb_base, Pn_base, Pm_base, inv, Rlen, Ra, Rb, Rm, Rn,
4041       Pa, Pb, Pn, Pm, Rhi_ab, Rlo_ab, Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2, Ri, Rj;
4042 
4043     RegSet _toSave;
4044     bool _squaring;
4045 
4046   public:
4047     MontgomeryMultiplyGenerator (Assembler *as, bool squaring)
4048       : MacroAssembler(as->code()), _squaring(squaring) {
4049 
4050       // Register allocation
4051 
4052       RegSetIterator<Register> regs = RegSet::range(x10, x26).begin();
4053       Pa_base = *regs;       // Argument registers
4054       if (squaring) {
4055         Pb_base = Pa_base;
4056       } else {
4057         Pb_base = *++regs;
4058       }
4059       Pn_base = *++regs;
4060       Rlen= *++regs;
4061       inv = *++regs;
4062       Pm_base = *++regs;
4063 
4064                         // Working registers:
4065       Ra =  *++regs;    // The current digit of a, b, n, and m.
4066       Rb =  *++regs;
4067       Rm =  *++regs;
4068       Rn =  *++regs;
4069 
4070       Pa =  *++regs;      // Pointers to the current/next digit of a, b, n, and m.
4071       Pb =  *++regs;
4072       Pm =  *++regs;
4073       Pn =  *++regs;
4074 
4075       tmp0 =  *++regs;    // Three registers which form a
4076       tmp1 =  *++regs;    // triple-precision accumuator.
4077       tmp2 =  *++regs;
4078 
4079       Ri =  x6;         // Inner and outer loop indexes.
4080       Rj =  x7;
4081 
4082       Rhi_ab = x28;     // Product registers: low and high parts
4083       Rlo_ab = x29;     // of a*b and m*n.
4084       Rhi_mn = x30;
4085       Rlo_mn = x31;
4086 
4087       // x18 and up are callee-saved.
4088       _toSave = RegSet::range(x18, *regs) + Pm_base;
4089     }
4090 
4091   private:
4092     void save_regs() {
4093       push_reg(_toSave, sp);
4094     }
4095 
4096     void restore_regs() {
4097       pop_reg(_toSave, sp);
4098     }
4099 
4100     template <typename T>
4101     void unroll_2(Register count, T block) {
4102       Label loop, end, odd;
4103       beqz(count, end);
4104       test_bit(t0, count, 0);
4105       bnez(t0, odd);
4106       align(16);
4107       bind(loop);
4108       (this->*block)();
4109       bind(odd);
4110       (this->*block)();
4111       subi(count, count, 2);
4112       bgtz(count, loop);
4113       bind(end);
4114     }
4115 
4116     template <typename T>
4117     void unroll_2(Register count, T block, Register d, Register s, Register tmp) {
4118       Label loop, end, odd;
4119       beqz(count, end);
4120       test_bit(tmp, count, 0);
4121       bnez(tmp, odd);
4122       align(16);
4123       bind(loop);
4124       (this->*block)(d, s, tmp);
4125       bind(odd);
4126       (this->*block)(d, s, tmp);
4127       subi(count, count, 2);
4128       bgtz(count, loop);
4129       bind(end);
4130     }
4131 
4132     void pre1(RegisterOrConstant i) {
4133       block_comment("pre1");
4134       // Pa = Pa_base;
4135       // Pb = Pb_base + i;
4136       // Pm = Pm_base;
4137       // Pn = Pn_base + i;
4138       // Ra = *Pa;
4139       // Rb = *Pb;
4140       // Rm = *Pm;
4141       // Rn = *Pn;
4142       if (i.is_register()) {
4143         slli(t0, i.as_register(), LogBytesPerWord);
4144       } else {
4145         mv(t0, i.as_constant());
4146         slli(t0, t0, LogBytesPerWord);
4147       }
4148 
4149       mv(Pa, Pa_base);
4150       add(Pb, Pb_base, t0);
4151       mv(Pm, Pm_base);
4152       add(Pn, Pn_base, t0);
4153 
4154       ld(Ra, Address(Pa));
4155       ld(Rb, Address(Pb));
4156       ld(Rm, Address(Pm));
4157       ld(Rn, Address(Pn));
4158 
4159       // Zero the m*n result.
4160       mv(Rhi_mn, zr);
4161       mv(Rlo_mn, zr);
4162     }
4163 
4164     // The core multiply-accumulate step of a Montgomery
4165     // multiplication.  The idea is to schedule operations as a
4166     // pipeline so that instructions with long latencies (loads and
4167     // multiplies) have time to complete before their results are
4168     // used.  This most benefits in-order implementations of the
4169     // architecture but out-of-order ones also benefit.
4170     void step() {
4171       block_comment("step");
4172       // MACC(Ra, Rb, tmp0, tmp1, tmp2);
4173       // Ra = *++Pa;
4174       // Rb = *--Pb;
4175       mulhu(Rhi_ab, Ra, Rb);
4176       mul(Rlo_ab, Ra, Rb);
4177       addi(Pa, Pa, wordSize);
4178       ld(Ra, Address(Pa));
4179       subi(Pb, Pb, wordSize);
4180       ld(Rb, Address(Pb));
4181       acc(Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2); // The pending m*n from the
4182                                             // previous iteration.
4183       // MACC(Rm, Rn, tmp0, tmp1, tmp2);
4184       // Rm = *++Pm;
4185       // Rn = *--Pn;
4186       mulhu(Rhi_mn, Rm, Rn);
4187       mul(Rlo_mn, Rm, Rn);
4188       addi(Pm, Pm, wordSize);
4189       ld(Rm, Address(Pm));
4190       subi(Pn, Pn, wordSize);
4191       ld(Rn, Address(Pn));
4192       acc(Rhi_ab, Rlo_ab, tmp0, tmp1, tmp2);
4193     }
4194 
4195     void post1() {
4196       block_comment("post1");
4197 
4198       // MACC(Ra, Rb, tmp0, tmp1, tmp2);
4199       // Ra = *++Pa;
4200       // Rb = *--Pb;
4201       mulhu(Rhi_ab, Ra, Rb);
4202       mul(Rlo_ab, Ra, Rb);
4203       acc(Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2);  // The pending m*n
4204       acc(Rhi_ab, Rlo_ab, tmp0, tmp1, tmp2);
4205 
4206       // *Pm = Rm = tmp0 * inv;
4207       mul(Rm, tmp0, inv);
4208       sd(Rm, Address(Pm));
4209 
4210       // MACC(Rm, Rn, tmp0, tmp1, tmp2);
4211       // tmp0 = tmp1; tmp1 = tmp2; tmp2 = 0;
4212       mulhu(Rhi_mn, Rm, Rn);
4213 
4214 #ifndef PRODUCT
4215       // assert(m[i] * n[0] + tmp0 == 0, "broken Montgomery multiply");
4216       {
4217         mul(Rlo_mn, Rm, Rn);
4218         add(Rlo_mn, tmp0, Rlo_mn);
4219         Label ok;
4220         beqz(Rlo_mn, ok);
4221         stop("broken Montgomery multiply");
4222         bind(ok);
4223       }
4224 #endif
4225       // We have very carefully set things up so that
4226       // m[i]*n[0] + tmp0 == 0 (mod b), so we don't have to calculate
4227       // the lower half of Rm * Rn because we know the result already:
4228       // it must be -tmp0.  tmp0 + (-tmp0) must generate a carry iff
4229       // tmp0 != 0.  So, rather than do a mul and an cad we just set
4230       // the carry flag iff tmp0 is nonzero.
4231       //
4232       // mul(Rlo_mn, Rm, Rn);
4233       // cad(zr, tmp0, Rlo_mn);
4234       subi(t0, tmp0, 1);
4235       sltu(t0, t0, tmp0); // Set carry iff tmp0 is nonzero
4236       cadc(tmp0, tmp1, Rhi_mn, t0);
4237       adc(tmp1, tmp2, zr, t0);
4238       mv(tmp2, zr);
4239     }
4240 
4241     void pre2(Register i, Register len) {
4242       block_comment("pre2");
4243       // Pa = Pa_base + i-len;
4244       // Pb = Pb_base + len;
4245       // Pm = Pm_base + i-len;
4246       // Pn = Pn_base + len;
4247 
4248       sub(Rj, i, len);
4249       // Rj == i-len
4250 
4251       // Ra as temp register
4252       slli(Ra, Rj, LogBytesPerWord);
4253       add(Pa, Pa_base, Ra);
4254       add(Pm, Pm_base, Ra);
4255       slli(Ra, len, LogBytesPerWord);
4256       add(Pb, Pb_base, Ra);
4257       add(Pn, Pn_base, Ra);
4258 
4259       // Ra = *++Pa;
4260       // Rb = *--Pb;
4261       // Rm = *++Pm;
4262       // Rn = *--Pn;
4263       addi(Pa, Pa, wordSize);
4264       ld(Ra, Address(Pa));
4265       subi(Pb, Pb, wordSize);
4266       ld(Rb, Address(Pb));
4267       addi(Pm, Pm, wordSize);
4268       ld(Rm, Address(Pm));
4269       subi(Pn, Pn, wordSize);
4270       ld(Rn, Address(Pn));
4271 
4272       mv(Rhi_mn, zr);
4273       mv(Rlo_mn, zr);
4274     }
4275 
4276     void post2(Register i, Register len) {
4277       block_comment("post2");
4278       sub(Rj, i, len);
4279 
4280       cad(tmp0, tmp0, Rlo_mn, t0); // The pending m*n, low part
4281 
4282       // As soon as we know the least significant digit of our result,
4283       // store it.
4284       // Pm_base[i-len] = tmp0;
4285       // Rj as temp register
4286       slli(Rj, Rj, LogBytesPerWord);
4287       add(Rj, Pm_base, Rj);
4288       sd(tmp0, Address(Rj));
4289 
4290       // tmp0 = tmp1; tmp1 = tmp2; tmp2 = 0;
4291       cadc(tmp0, tmp1, Rhi_mn, t0); // The pending m*n, high part
4292       adc(tmp1, tmp2, zr, t0);
4293       mv(tmp2, zr);
4294     }
4295 
4296     // A carry in tmp0 after Montgomery multiplication means that we
4297     // should subtract multiples of n from our result in m.  We'll
4298     // keep doing that until there is no carry.
4299     void normalize(Register len) {
4300       block_comment("normalize");
4301       // while (tmp0)
4302       //   tmp0 = sub(Pm_base, Pn_base, tmp0, len);
4303       Label loop, post, again;
4304       Register cnt = tmp1, i = tmp2; // Re-use registers; we're done with them now
4305       beqz(tmp0, post); {
4306         bind(again); {
4307           mv(i, zr);
4308           mv(cnt, len);
4309           slli(Rn, i, LogBytesPerWord);
4310           add(Rm, Pm_base, Rn);
4311           ld(Rm, Address(Rm));
4312           add(Rn, Pn_base, Rn);
4313           ld(Rn, Address(Rn));
4314           mv(t0, 1); // set carry flag, i.e. no borrow
4315           align(16);
4316           bind(loop); {
4317             notr(Rn, Rn);
4318             add(Rm, Rm, t0);
4319             add(Rm, Rm, Rn);
4320             sltu(t0, Rm, Rn);
4321             slli(Rn, i, LogBytesPerWord); // Rn as temp register
4322             add(Rn, Pm_base, Rn);
4323             sd(Rm, Address(Rn));
4324             addi(i, i, 1);
4325             slli(Rn, i, LogBytesPerWord);
4326             add(Rm, Pm_base, Rn);
4327             ld(Rm, Address(Rm));
4328             add(Rn, Pn_base, Rn);
4329             ld(Rn, Address(Rn));
4330             subi(cnt, cnt, 1);
4331           } bnez(cnt, loop);
4332           subi(tmp0, tmp0, 1);
4333           add(tmp0, tmp0, t0);
4334         } bnez(tmp0, again);
4335       } bind(post);
4336     }
4337 
4338     // Move memory at s to d, reversing words.
4339     //    Increments d to end of copied memory
4340     //    Destroys tmp1, tmp2
4341     //    Preserves len
4342     //    Leaves s pointing to the address which was in d at start
4343     void reverse(Register d, Register s, Register len, Register tmp1, Register tmp2) {
4344       assert(tmp1->encoding() < x28->encoding(), "register corruption");
4345       assert(tmp2->encoding() < x28->encoding(), "register corruption");
4346 
4347       shadd(s, len, s, tmp1, LogBytesPerWord);
4348       mv(tmp1, len);
4349       unroll_2(tmp1,  &MontgomeryMultiplyGenerator::reverse1, d, s, tmp2);
4350       slli(tmp1, len, LogBytesPerWord);
4351       sub(s, d, tmp1);
4352     }
4353     // [63...0] -> [31...0][63...32]
4354     void reverse1(Register d, Register s, Register tmp) {
4355       subi(s, s, wordSize);
4356       ld(tmp, Address(s));
4357       ror(tmp, tmp, 32, t0);
4358       sd(tmp, Address(d));
4359       addi(d, d, wordSize);
4360     }
4361 
4362     void step_squaring() {
4363       // An extra ACC
4364       step();
4365       acc(Rhi_ab, Rlo_ab, tmp0, tmp1, tmp2);
4366     }
4367 
4368     void last_squaring(Register i) {
4369       Label dont;
4370       // if ((i & 1) == 0) {
4371       test_bit(t0, i, 0);
4372       bnez(t0, dont); {
4373         // MACC(Ra, Rb, tmp0, tmp1, tmp2);
4374         // Ra = *++Pa;
4375         // Rb = *--Pb;
4376         mulhu(Rhi_ab, Ra, Rb);
4377         mul(Rlo_ab, Ra, Rb);
4378         acc(Rhi_ab, Rlo_ab, tmp0, tmp1, tmp2);
4379       } bind(dont);
4380     }
4381 
4382     void extra_step_squaring() {
4383       acc(Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2);  // The pending m*n
4384 
4385       // MACC(Rm, Rn, tmp0, tmp1, tmp2);
4386       // Rm = *++Pm;
4387       // Rn = *--Pn;
4388       mulhu(Rhi_mn, Rm, Rn);
4389       mul(Rlo_mn, Rm, Rn);
4390       addi(Pm, Pm, wordSize);
4391       ld(Rm, Address(Pm));
4392       subi(Pn, Pn, wordSize);
4393       ld(Rn, Address(Pn));
4394     }
4395 
4396     void post1_squaring() {
4397       acc(Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2);  // The pending m*n
4398 
4399       // *Pm = Rm = tmp0 * inv;
4400       mul(Rm, tmp0, inv);
4401       sd(Rm, Address(Pm));
4402 
4403       // MACC(Rm, Rn, tmp0, tmp1, tmp2);
4404       // tmp0 = tmp1; tmp1 = tmp2; tmp2 = 0;
4405       mulhu(Rhi_mn, Rm, Rn);
4406 
4407 #ifndef PRODUCT
4408       // assert(m[i] * n[0] + tmp0 == 0, "broken Montgomery multiply");
4409       {
4410         mul(Rlo_mn, Rm, Rn);
4411         add(Rlo_mn, tmp0, Rlo_mn);
4412         Label ok;
4413         beqz(Rlo_mn, ok); {
4414           stop("broken Montgomery multiply");
4415         } bind(ok);
4416       }
4417 #endif
4418       // We have very carefully set things up so that
4419       // m[i]*n[0] + tmp0 == 0 (mod b), so we don't have to calculate
4420       // the lower half of Rm * Rn because we know the result already:
4421       // it must be -tmp0.  tmp0 + (-tmp0) must generate a carry iff
4422       // tmp0 != 0.  So, rather than do a mul and a cad we just set
4423       // the carry flag iff tmp0 is nonzero.
4424       //
4425       // mul(Rlo_mn, Rm, Rn);
4426       // cad(zr, tmp, Rlo_mn);
4427       subi(t0, tmp0, 1);
4428       sltu(t0, t0, tmp0); // Set carry iff tmp0 is nonzero
4429       cadc(tmp0, tmp1, Rhi_mn, t0);
4430       adc(tmp1, tmp2, zr, t0);
4431       mv(tmp2, zr);
4432     }
4433 
4434     // use t0 as carry
4435     void acc(Register Rhi, Register Rlo,
4436              Register tmp0, Register tmp1, Register tmp2) {
4437       cad(tmp0, tmp0, Rlo, t0);
4438       cadc(tmp1, tmp1, Rhi, t0);
4439       adc(tmp2, tmp2, zr, t0);
4440     }
4441 
4442   public:
4443     /**
4444      * Fast Montgomery multiplication.  The derivation of the
4445      * algorithm is in A Cryptographic Library for the Motorola
4446      * DSP56000, Dusse and Kaliski, Proc. EUROCRYPT 90, pp. 230-237.
4447      *
4448      * Arguments:
4449      *
4450      * Inputs for multiplication:
4451      *   c_rarg0   - int array elements a
4452      *   c_rarg1   - int array elements b
4453      *   c_rarg2   - int array elements n (the modulus)
4454      *   c_rarg3   - int length
4455      *   c_rarg4   - int inv
4456      *   c_rarg5   - int array elements m (the result)
4457      *
4458      * Inputs for squaring:
4459      *   c_rarg0   - int array elements a
4460      *   c_rarg1   - int array elements n (the modulus)
4461      *   c_rarg2   - int length
4462      *   c_rarg3   - int inv
4463      *   c_rarg4   - int array elements m (the result)
4464      *
4465      */
4466     address generate_multiply() {
4467       Label argh, nothing;
4468       bind(argh);
4469       stop("MontgomeryMultiply total_allocation must be <= 8192");
4470 
4471       align(CodeEntryAlignment);
4472       address entry = pc();
4473 
4474       beqz(Rlen, nothing);
4475 
4476       enter();
4477 
4478       // Make room.
4479       mv(Ra, 512);
4480       bgt(Rlen, Ra, argh);
4481       slli(Ra, Rlen, exact_log2(4 * sizeof(jint)));
4482       sub(Ra, sp, Ra);
4483       andi(sp, Ra, -2 * wordSize);
4484 
4485       srliw(Rlen, Rlen, 1);  // length in longwords = len/2
4486 
4487       {
4488         // Copy input args, reversing as we go.  We use Ra as a
4489         // temporary variable.
4490         reverse(Ra, Pa_base, Rlen, Ri, Rj);
4491         if (!_squaring)
4492           reverse(Ra, Pb_base, Rlen, Ri, Rj);
4493         reverse(Ra, Pn_base, Rlen, Ri, Rj);
4494       }
4495 
4496       // Push all call-saved registers and also Pm_base which we'll need
4497       // at the end.
4498       save_regs();
4499 
4500 #ifndef PRODUCT
4501       // assert(inv * n[0] == -1UL, "broken inverse in Montgomery multiply");
4502       {
4503         ld(Rn, Address(Pn_base));
4504         mul(Rlo_mn, Rn, inv);
4505         mv(t0, -1);
4506         Label ok;
4507         beq(Rlo_mn, t0, ok);
4508         stop("broken inverse in Montgomery multiply");
4509         bind(ok);
4510       }
4511 #endif
4512 
4513       mv(Pm_base, Ra);
4514 
4515       mv(tmp0, zr);
4516       mv(tmp1, zr);
4517       mv(tmp2, zr);
4518 
4519       block_comment("for (int i = 0; i < len; i++) {");
4520       mv(Ri, zr); {
4521         Label loop, end;
4522         bge(Ri, Rlen, end);
4523 
4524         bind(loop);
4525         pre1(Ri);
4526 
4527         block_comment("  for (j = i; j; j--) {"); {
4528           mv(Rj, Ri);
4529           unroll_2(Rj, &MontgomeryMultiplyGenerator::step);
4530         } block_comment("  } // j");
4531 
4532         post1();
4533         addiw(Ri, Ri, 1);
4534         blt(Ri, Rlen, loop);
4535         bind(end);
4536         block_comment("} // i");
4537       }
4538 
4539       block_comment("for (int i = len; i < 2*len; i++) {");
4540       mv(Ri, Rlen); {
4541         Label loop, end;
4542         slli(t0, Rlen, 1);
4543         bge(Ri, t0, end);
4544 
4545         bind(loop);
4546         pre2(Ri, Rlen);
4547 
4548         block_comment("  for (j = len*2-i-1; j; j--) {"); {
4549           slliw(Rj, Rlen, 1);
4550           subw(Rj, Rj, Ri);
4551           subiw(Rj, Rj, 1);
4552           unroll_2(Rj, &MontgomeryMultiplyGenerator::step);
4553         } block_comment("  } // j");
4554 
4555         post2(Ri, Rlen);
4556         addiw(Ri, Ri, 1);
4557         slli(t0, Rlen, 1);
4558         blt(Ri, t0, loop);
4559         bind(end);
4560       }
4561       block_comment("} // i");
4562 
4563       normalize(Rlen);
4564 
4565       mv(Ra, Pm_base);  // Save Pm_base in Ra
4566       restore_regs();  // Restore caller's Pm_base
4567 
4568       // Copy our result into caller's Pm_base
4569       reverse(Pm_base, Ra, Rlen, Ri, Rj);
4570 
4571       leave();
4572       bind(nothing);
4573       ret();
4574 
4575       return entry;
4576     }
4577 
4578     /**
4579      *
4580      * Arguments:
4581      *
4582      * Inputs:
4583      *   c_rarg0   - int array elements a
4584      *   c_rarg1   - int array elements n (the modulus)
4585      *   c_rarg2   - int length
4586      *   c_rarg3   - int inv
4587      *   c_rarg4   - int array elements m (the result)
4588      *
4589      */
4590     address generate_square() {
4591       Label argh;
4592       bind(argh);
4593       stop("MontgomeryMultiply total_allocation must be <= 8192");
4594 
4595       align(CodeEntryAlignment);
4596       address entry = pc();
4597 
4598       enter();
4599 
4600       // Make room.
4601       mv(Ra, 512);
4602       bgt(Rlen, Ra, argh);
4603       slli(Ra, Rlen, exact_log2(4 * sizeof(jint)));
4604       sub(Ra, sp, Ra);
4605       andi(sp, Ra, -2 * wordSize);
4606 
4607       srliw(Rlen, Rlen, 1);  // length in longwords = len/2
4608 
4609       {
4610         // Copy input args, reversing as we go.  We use Ra as a
4611         // temporary variable.
4612         reverse(Ra, Pa_base, Rlen, Ri, Rj);
4613         reverse(Ra, Pn_base, Rlen, Ri, Rj);
4614       }
4615 
4616       // Push all call-saved registers and also Pm_base which we'll need
4617       // at the end.
4618       save_regs();
4619 
4620       mv(Pm_base, Ra);
4621 
4622       mv(tmp0, zr);
4623       mv(tmp1, zr);
4624       mv(tmp2, zr);
4625 
4626       block_comment("for (int i = 0; i < len; i++) {");
4627       mv(Ri, zr); {
4628         Label loop, end;
4629         bind(loop);
4630         bge(Ri, Rlen, end);
4631 
4632         pre1(Ri);
4633 
4634         block_comment("for (j = (i+1)/2; j; j--) {"); {
4635           addi(Rj, Ri, 1);
4636           srliw(Rj, Rj, 1);
4637           unroll_2(Rj, &MontgomeryMultiplyGenerator::step_squaring);
4638         } block_comment("  } // j");
4639 
4640         last_squaring(Ri);
4641 
4642         block_comment("  for (j = i/2; j; j--) {"); {
4643           srliw(Rj, Ri, 1);
4644           unroll_2(Rj, &MontgomeryMultiplyGenerator::extra_step_squaring);
4645         } block_comment("  } // j");
4646 
4647         post1_squaring();
4648         addi(Ri, Ri, 1);
4649         blt(Ri, Rlen, loop);
4650 
4651         bind(end);
4652         block_comment("} // i");
4653       }
4654 
4655       block_comment("for (int i = len; i < 2*len; i++) {");
4656       mv(Ri, Rlen); {
4657         Label loop, end;
4658         bind(loop);
4659         slli(t0, Rlen, 1);
4660         bge(Ri, t0, end);
4661 
4662         pre2(Ri, Rlen);
4663 
4664         block_comment("  for (j = (2*len-i-1)/2; j; j--) {"); {
4665           slli(Rj, Rlen, 1);
4666           sub(Rj, Rj, Ri);
4667           subi(Rj, Rj, 1);
4668           srliw(Rj, Rj, 1);
4669           unroll_2(Rj, &MontgomeryMultiplyGenerator::step_squaring);
4670         } block_comment("  } // j");
4671 
4672         last_squaring(Ri);
4673 
4674         block_comment("  for (j = (2*len-i)/2; j; j--) {"); {
4675           slli(Rj, Rlen, 1);
4676           sub(Rj, Rj, Ri);
4677           srliw(Rj, Rj, 1);
4678           unroll_2(Rj, &MontgomeryMultiplyGenerator::extra_step_squaring);
4679         } block_comment("  } // j");
4680 
4681         post2(Ri, Rlen);
4682         addi(Ri, Ri, 1);
4683         slli(t0, Rlen, 1);
4684         blt(Ri, t0, loop);
4685 
4686         bind(end);
4687         block_comment("} // i");
4688       }
4689 
4690       normalize(Rlen);
4691 
4692       mv(Ra, Pm_base);  // Save Pm_base in Ra
4693       restore_regs();  // Restore caller's Pm_base
4694 
4695       // Copy our result into caller's Pm_base
4696       reverse(Pm_base, Ra, Rlen, Ri, Rj);
4697 
4698       leave();
4699       ret();
4700 
4701       return entry;
4702     }
4703   };
4704 
4705 #endif // COMPILER2
4706 
4707   address generate_cont_thaw(Continuation::thaw_kind kind) {
4708     bool return_barrier = Continuation::is_thaw_return_barrier(kind);
4709     bool return_barrier_exception = Continuation::is_thaw_return_barrier_exception(kind);
4710 
4711     address start = __ pc();
4712 
4713     if (return_barrier) {
4714       __ ld(sp, Address(xthread, JavaThread::cont_entry_offset()));
4715     }
4716 
4717 #ifndef PRODUCT
4718     {
4719       Label OK;
4720       __ ld(t0, Address(xthread, JavaThread::cont_entry_offset()));
4721       __ beq(sp, t0, OK);
4722       __ stop("incorrect sp");
4723       __ bind(OK);
4724     }
4725 #endif
4726 
4727     if (return_barrier) {
4728       // preserve possible return value from a method returning to the return barrier
4729       __ subi(sp, sp, 2 * wordSize);
4730       __ fsd(f10, Address(sp, 0 * wordSize));
4731       __ sd(x10, Address(sp, 1 * wordSize));
4732     }
4733 
4734     __ mv(c_rarg1, (return_barrier ? 1 : 0));
4735     __ call_VM_leaf(CAST_FROM_FN_PTR(address, Continuation::prepare_thaw), xthread, c_rarg1);
4736     __ mv(t1, x10); // x10 contains the size of the frames to thaw, 0 if overflow or no more frames
4737 
4738     if (return_barrier) {
4739       // restore return value (no safepoint in the call to thaw, so even an oop return value should be OK)
4740       __ ld(x10, Address(sp, 1 * wordSize));
4741       __ fld(f10, Address(sp, 0 * wordSize));
4742       __ addi(sp, sp, 2 * wordSize);
4743     }
4744 
4745 #ifndef PRODUCT
4746     {
4747       Label OK;
4748       __ ld(t0, Address(xthread, JavaThread::cont_entry_offset()));
4749       __ beq(sp, t0, OK);
4750       __ stop("incorrect sp");
4751       __ bind(OK);
4752     }
4753 #endif
4754 
4755     Label thaw_success;
4756     // t1 contains the size of the frames to thaw, 0 if overflow or no more frames
4757     __ bnez(t1, thaw_success);
4758     __ j(RuntimeAddress(SharedRuntime::throw_StackOverflowError_entry()));
4759     __ bind(thaw_success);
4760 
4761     // make room for the thawed frames
4762     __ sub(t0, sp, t1);
4763     __ andi(sp, t0, -16); // align
4764 
4765     if (return_barrier) {
4766       // save original return value -- again
4767       __ subi(sp, sp, 2 * wordSize);
4768       __ fsd(f10, Address(sp, 0 * wordSize));
4769       __ sd(x10, Address(sp, 1 * wordSize));
4770     }
4771 
4772     // If we want, we can templatize thaw by kind, and have three different entries
4773     __ mv(c_rarg1, kind);
4774 
4775     __ call_VM_leaf(Continuation::thaw_entry(), xthread, c_rarg1);
4776     __ mv(t1, x10); // x10 is the sp of the yielding frame
4777 
4778     if (return_barrier) {
4779       // restore return value (no safepoint in the call to thaw, so even an oop return value should be OK)
4780       __ ld(x10, Address(sp, 1 * wordSize));
4781       __ fld(f10, Address(sp, 0 * wordSize));
4782       __ addi(sp, sp, 2 * wordSize);
4783     } else {
4784       __ mv(x10, zr); // return 0 (success) from doYield
4785     }
4786 
4787     // we're now on the yield frame (which is in an address above us b/c sp has been pushed down)
4788     __ mv(fp, t1);
4789     __ subi(sp, t1, 2 * wordSize); // now pointing to fp spill
4790 
4791     if (return_barrier_exception) {
4792       __ ld(c_rarg1, Address(fp, -1 * wordSize)); // return address
4793       __ verify_oop(x10);
4794       __ mv(x9, x10); // save return value contaning the exception oop in callee-saved x9
4795 
4796       __ call_VM_leaf(CAST_FROM_FN_PTR(address, SharedRuntime::exception_handler_for_return_address), xthread, c_rarg1);
4797 
4798       // see OptoRuntime::generate_exception_blob: x10 -- exception oop, x13 -- exception pc
4799 
4800       __ mv(x11, x10); // the exception handler
4801       __ mv(x10, x9); // restore return value contaning the exception oop
4802       __ verify_oop(x10);
4803 
4804       __ leave();
4805       __ mv(x13, ra);
4806       __ jr(x11); // the exception handler
4807     } else {
4808       // We're "returning" into the topmost thawed frame; see Thaw::push_return_frame
4809       __ leave();
4810       __ ret();
4811     }
4812 
4813     return start;
4814   }
4815 
4816   address generate_cont_thaw() {
4817     if (!Continuations::enabled()) return nullptr;
4818 
4819     StubId stub_id = StubId::stubgen_cont_thaw_id;
4820     StubCodeMark mark(this, stub_id);
4821     address start = __ pc();
4822     generate_cont_thaw(Continuation::thaw_top);
4823     return start;
4824   }
4825 
4826   address generate_cont_returnBarrier() {
4827     if (!Continuations::enabled()) return nullptr;
4828 
4829     // TODO: will probably need multiple return barriers depending on return type
4830     StubId stub_id = StubId::stubgen_cont_returnBarrier_id;
4831     StubCodeMark mark(this, stub_id);
4832     address start = __ pc();
4833 
4834     generate_cont_thaw(Continuation::thaw_return_barrier);
4835 
4836     return start;
4837   }
4838 
4839   address generate_cont_returnBarrier_exception() {
4840     if (!Continuations::enabled()) return nullptr;
4841 
4842     StubId stub_id = StubId::stubgen_cont_returnBarrierExc_id;
4843     StubCodeMark mark(this, stub_id);
4844     address start = __ pc();
4845 
4846     generate_cont_thaw(Continuation::thaw_return_barrier_exception);
4847 
4848     return start;
4849   }
4850 
4851   address generate_cont_preempt_stub() {
4852     if (!Continuations::enabled()) return nullptr;
4853     StubId stub_id = StubId::stubgen_cont_preempt_id;
4854     StubCodeMark mark(this, stub_id);
4855     address start = __ pc();
4856 
4857     __ reset_last_Java_frame(true);
4858 
4859     // Set sp to enterSpecial frame, i.e. remove all frames copied into the heap.
4860     __ ld(sp, Address(xthread, JavaThread::cont_entry_offset()));
4861 
4862     Label preemption_cancelled;
4863     __ lbu(t0, Address(xthread, JavaThread::preemption_cancelled_offset()));
4864     __ bnez(t0, preemption_cancelled);
4865 
4866     // Remove enterSpecial frame from the stack and return to Continuation.run() to unmount.
4867     SharedRuntime::continuation_enter_cleanup(_masm);
4868     __ leave();
4869     __ ret();
4870 
4871     // We acquired the monitor after freezing the frames so call thaw to continue execution.
4872     __ bind(preemption_cancelled);
4873     __ sb(zr, Address(xthread, JavaThread::preemption_cancelled_offset()));
4874     __ la(fp, Address(sp, checked_cast<int32_t>(ContinuationEntry::size() + 2 * wordSize)));
4875     __ la(t1, ExternalAddress(ContinuationEntry::thaw_call_pc_address()));
4876     __ ld(t1, Address(t1));
4877     __ jr(t1);
4878 
4879     return start;
4880   }
4881 
4882 #ifdef COMPILER2
4883 
4884 #undef __
4885 #define __ this->
4886 
4887   class Sha2Generator : public MacroAssembler {
4888     StubCodeGenerator* _cgen;
4889    public:
4890       Sha2Generator(MacroAssembler* masm, StubCodeGenerator* cgen) : MacroAssembler(masm->code()), _cgen(cgen) {}
4891       address generate_sha256_implCompress(StubId stub_id) {
4892         return generate_sha2_implCompress(Assembler::e32, stub_id);
4893       }
4894       address generate_sha512_implCompress(StubId stub_id) {
4895         return generate_sha2_implCompress(Assembler::e64, stub_id);
4896       }
4897    private:
4898 
4899     void vleXX_v(Assembler::SEW vset_sew, VectorRegister vr, Register sr) {
4900       if (vset_sew == Assembler::e32) __ vle32_v(vr, sr);
4901       else                            __ vle64_v(vr, sr);
4902     }
4903 
4904     void vseXX_v(Assembler::SEW vset_sew, VectorRegister vr, Register sr) {
4905       if (vset_sew == Assembler::e32) __ vse32_v(vr, sr);
4906       else                            __ vse64_v(vr, sr);
4907     }
4908 
4909     // Overview of the logic in each "quad round".
4910     //
4911     // The code below repeats 16/20 times the logic implementing four rounds
4912     // of the SHA-256/512 core loop as documented by NIST. 16/20 "quad rounds"
4913     // to implementing the 64/80 single rounds.
4914     //
4915     //    // Load four word (u32/64) constants (K[t+3], K[t+2], K[t+1], K[t+0])
4916     //    // Output:
4917     //    //   vTmp1 = {K[t+3], K[t+2], K[t+1], K[t+0]}
4918     //    vl1reXX.v vTmp1, ofs
4919     //
4920     //    // Increment word constant address by stride (16/32 bytes, 4*4B/8B, 128b/256b)
4921     //    addi ofs, ofs, 16/32
4922     //
4923     //    // Add constants to message schedule words:
4924     //    //  Input
4925     //    //    vTmp1 = {K[t+3], K[t+2], K[t+1], K[t+0]}
4926     //    //    vW0 = {W[t+3], W[t+2], W[t+1], W[t+0]}; // Vt0 = W[3:0];
4927     //    //  Output
4928     //    //    vTmp0 = {W[t+3]+K[t+3], W[t+2]+K[t+2], W[t+1]+K[t+1], W[t+0]+K[t+0]}
4929     //    vadd.vv vTmp0, vTmp1, vW0
4930     //
4931     //    //  2 rounds of working variables updates.
4932     //    //     vState1[t+4] <- vState1[t], vState0[t], vTmp0[t]
4933     //    //  Input:
4934     //    //    vState1 = {c[t],d[t],g[t],h[t]}   " = vState1[t] "
4935     //    //    vState0 = {a[t],b[t],e[t],f[t]}
4936     //    //    vTmp0 = {W[t+3]+K[t+3], W[t+2]+K[t+2], W[t+1]+K[t+1], W[t+0]+K[t+0]}
4937     //    //  Output:
4938     //    //    vState1 = {f[t+2],e[t+2],b[t+2],a[t+2]}  " = vState0[t+2] "
4939     //    //        = {h[t+4],g[t+4],d[t+4],c[t+4]}  " = vState1[t+4] "
4940     //    vsha2cl.vv vState1, vState0, vTmp0
4941     //
4942     //    //  2 rounds of working variables updates.
4943     //    //     vState0[t+4] <- vState0[t], vState0[t+2], vTmp0[t]
4944     //    //  Input
4945     //    //   vState0 = {a[t],b[t],e[t],f[t]}       " = vState0[t] "
4946     //    //       = {h[t+2],g[t+2],d[t+2],c[t+2]}   " = vState1[t+2] "
4947     //    //   vState1 = {f[t+2],e[t+2],b[t+2],a[t+2]}   " = vState0[t+2] "
4948     //    //   vTmp0 = {W[t+3]+K[t+3], W[t+2]+K[t+2], W[t+1]+K[t+1], W[t+0]+K[t+0]}
4949     //    //  Output:
4950     //    //   vState0 = {f[t+4],e[t+4],b[t+4],a[t+4]}   " = vState0[t+4] "
4951     //    vsha2ch.vv vState0, vState1, vTmp0
4952     //
4953     //    // Combine 2QW into 1QW
4954     //    //
4955     //    // To generate the next 4 words, "new_vW0"/"vTmp0" from vW0-vW3, vsha2ms needs
4956     //    //     vW0[0..3], vW1[0], vW2[1..3], vW3[0, 2..3]
4957     //    // and it can only take 3 vectors as inputs. Hence we need to combine
4958     //    // vW1[0] and vW2[1..3] in a single vector.
4959     //    //
4960     //    // vmerge Vt4, Vt1, Vt2, V0
4961     //    // Input
4962     //    //  V0 = mask // first word from vW2, 1..3 words from vW1
4963     //    //  vW2 = {Wt-8, Wt-7, Wt-6, Wt-5}
4964     //    //  vW1 = {Wt-12, Wt-11, Wt-10, Wt-9}
4965     //    // Output
4966     //    //  Vt4 = {Wt-12, Wt-7, Wt-6, Wt-5}
4967     //    vmerge.vvm vTmp0, vW2, vW1, v0
4968     //
4969     //    // Generate next Four Message Schedule Words (hence allowing for 4 more rounds)
4970     //    // Input
4971     //    //  vW0 = {W[t+ 3], W[t+ 2], W[t+ 1], W[t+ 0]}     W[ 3: 0]
4972     //    //  vW3 = {W[t+15], W[t+14], W[t+13], W[t+12]}     W[15:12]
4973     //    //  vTmp0 = {W[t+11], W[t+10], W[t+ 9], W[t+ 4]}     W[11: 9,4]
4974     //    // Output (next four message schedule words)
4975     //    //  vW0 = {W[t+19],  W[t+18],  W[t+17],  W[t+16]}  W[19:16]
4976     //    vsha2ms.vv vW0, vTmp0, vW3
4977     //
4978     // BEFORE
4979     //  vW0 - vW3 hold the message schedule words (initially the block words)
4980     //    vW0 = W[ 3: 0]   "oldest"
4981     //    vW1 = W[ 7: 4]
4982     //    vW2 = W[11: 8]
4983     //    vW3 = W[15:12]   "newest"
4984     //
4985     //  vt6 - vt7 hold the working state variables
4986     //    vState0 = {a[t],b[t],e[t],f[t]}   // initially {H5,H4,H1,H0}
4987     //    vState1 = {c[t],d[t],g[t],h[t]}   // initially {H7,H6,H3,H2}
4988     //
4989     // AFTER
4990     //  vW0 - vW3 hold the message schedule words (initially the block words)
4991     //    vW1 = W[ 7: 4]   "oldest"
4992     //    vW2 = W[11: 8]
4993     //    vW3 = W[15:12]
4994     //    vW0 = W[19:16]   "newest"
4995     //
4996     //  vState0 and vState1 hold the working state variables
4997     //    vState0 = {a[t+4],b[t+4],e[t+4],f[t+4]}
4998     //    vState1 = {c[t+4],d[t+4],g[t+4],h[t+4]}
4999     //
5000     //  The group of vectors vW0,vW1,vW2,vW3 is "rotated" by one in each quad-round,
5001     //  hence the uses of those vectors rotate in each round, and we get back to the
5002     //  initial configuration every 4 quad-rounds. We could avoid those changes at
5003     //  the cost of moving those vectors at the end of each quad-rounds.
5004     void sha2_quad_round(Assembler::SEW vset_sew, VectorRegister rot1, VectorRegister rot2, VectorRegister rot3, VectorRegister rot4,
5005                          Register scalarconst, VectorRegister vtemp, VectorRegister vtemp2, VectorRegister v_abef, VectorRegister v_cdgh,
5006                          bool gen_words = true, bool step_const = true) {
5007       __ vleXX_v(vset_sew, vtemp, scalarconst);
5008       if (step_const) {
5009         __ addi(scalarconst, scalarconst, vset_sew == Assembler::e32 ? 16 : 32);
5010       }
5011       __ vadd_vv(vtemp2, vtemp, rot1);
5012       __ vsha2cl_vv(v_cdgh, v_abef, vtemp2);
5013       __ vsha2ch_vv(v_abef, v_cdgh, vtemp2);
5014       if (gen_words) {
5015         __ vmerge_vvm(vtemp2, rot3, rot2);
5016         __ vsha2ms_vv(rot1, vtemp2, rot4);
5017       }
5018     }
5019 
5020     // Arguments:
5021     //
5022     // Inputs:
5023     //   c_rarg0   - byte[]  source+offset
5024     //   c_rarg1   - int[]   SHA.state
5025     //   c_rarg2   - int     offset
5026     //   c_rarg3   - int     limit
5027     //
5028     address generate_sha2_implCompress(Assembler::SEW vset_sew, StubId stub_id) {
5029       alignas(64) static const uint32_t round_consts_256[64] = {
5030         0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5,
5031         0x3956c25b, 0x59f111f1, 0x923f82a4, 0xab1c5ed5,
5032         0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3,
5033         0x72be5d74, 0x80deb1fe, 0x9bdc06a7, 0xc19bf174,
5034         0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc,
5035         0x2de92c6f, 0x4a7484aa, 0x5cb0a9dc, 0x76f988da,
5036         0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7,
5037         0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967,
5038         0x27b70a85, 0x2e1b2138, 0x4d2c6dfc, 0x53380d13,
5039         0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85,
5040         0xa2bfe8a1, 0xa81a664b, 0xc24b8b70, 0xc76c51a3,
5041         0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070,
5042         0x19a4c116, 0x1e376c08, 0x2748774c, 0x34b0bcb5,
5043         0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3,
5044         0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208,
5045         0x90befffa, 0xa4506ceb, 0xbef9a3f7, 0xc67178f2,
5046       };
5047       alignas(64) static const uint64_t round_consts_512[80] = {
5048         0x428a2f98d728ae22l, 0x7137449123ef65cdl, 0xb5c0fbcfec4d3b2fl,
5049         0xe9b5dba58189dbbcl, 0x3956c25bf348b538l, 0x59f111f1b605d019l,
5050         0x923f82a4af194f9bl, 0xab1c5ed5da6d8118l, 0xd807aa98a3030242l,
5051         0x12835b0145706fbel, 0x243185be4ee4b28cl, 0x550c7dc3d5ffb4e2l,
5052         0x72be5d74f27b896fl, 0x80deb1fe3b1696b1l, 0x9bdc06a725c71235l,
5053         0xc19bf174cf692694l, 0xe49b69c19ef14ad2l, 0xefbe4786384f25e3l,
5054         0x0fc19dc68b8cd5b5l, 0x240ca1cc77ac9c65l, 0x2de92c6f592b0275l,
5055         0x4a7484aa6ea6e483l, 0x5cb0a9dcbd41fbd4l, 0x76f988da831153b5l,
5056         0x983e5152ee66dfabl, 0xa831c66d2db43210l, 0xb00327c898fb213fl,
5057         0xbf597fc7beef0ee4l, 0xc6e00bf33da88fc2l, 0xd5a79147930aa725l,
5058         0x06ca6351e003826fl, 0x142929670a0e6e70l, 0x27b70a8546d22ffcl,
5059         0x2e1b21385c26c926l, 0x4d2c6dfc5ac42aedl, 0x53380d139d95b3dfl,
5060         0x650a73548baf63del, 0x766a0abb3c77b2a8l, 0x81c2c92e47edaee6l,
5061         0x92722c851482353bl, 0xa2bfe8a14cf10364l, 0xa81a664bbc423001l,
5062         0xc24b8b70d0f89791l, 0xc76c51a30654be30l, 0xd192e819d6ef5218l,
5063         0xd69906245565a910l, 0xf40e35855771202al, 0x106aa07032bbd1b8l,
5064         0x19a4c116b8d2d0c8l, 0x1e376c085141ab53l, 0x2748774cdf8eeb99l,
5065         0x34b0bcb5e19b48a8l, 0x391c0cb3c5c95a63l, 0x4ed8aa4ae3418acbl,
5066         0x5b9cca4f7763e373l, 0x682e6ff3d6b2b8a3l, 0x748f82ee5defb2fcl,
5067         0x78a5636f43172f60l, 0x84c87814a1f0ab72l, 0x8cc702081a6439ecl,
5068         0x90befffa23631e28l, 0xa4506cebde82bde9l, 0xbef9a3f7b2c67915l,
5069         0xc67178f2e372532bl, 0xca273eceea26619cl, 0xd186b8c721c0c207l,
5070         0xeada7dd6cde0eb1el, 0xf57d4f7fee6ed178l, 0x06f067aa72176fbal,
5071         0x0a637dc5a2c898a6l, 0x113f9804bef90dael, 0x1b710b35131c471bl,
5072         0x28db77f523047d84l, 0x32caab7b40c72493l, 0x3c9ebe0a15c9bebcl,
5073         0x431d67c49c100d4cl, 0x4cc5d4becb3e42b6l, 0x597f299cfc657e2al,
5074         0x5fcb6fab3ad6faecl, 0x6c44198c4a475817l
5075       };
5076       const int const_add = vset_sew == Assembler::e32 ? 16 : 32;
5077 
5078       bool multi_block;
5079       switch (stub_id) {
5080       case StubId::stubgen_sha256_implCompress_id:
5081         assert (vset_sew == Assembler::e32, "wrong macroassembler for stub");
5082         multi_block = false;
5083         break;
5084       case StubId::stubgen_sha256_implCompressMB_id:
5085         assert (vset_sew == Assembler::e32, "wrong macroassembler for stub");
5086         multi_block = true;
5087         break;
5088       case StubId::stubgen_sha512_implCompress_id:
5089         assert (vset_sew == Assembler::e64, "wrong macroassembler for stub");
5090         multi_block = false;
5091         break;
5092       case StubId::stubgen_sha512_implCompressMB_id:
5093         assert (vset_sew == Assembler::e64, "wrong macroassembler for stub");
5094         multi_block = true;
5095         break;
5096       default:
5097         ShouldNotReachHere();
5098       };
5099       __ align(CodeEntryAlignment);
5100       StubCodeMark mark(_cgen, stub_id);
5101       address start = __ pc();
5102 
5103       Register buf   = c_rarg0;
5104       Register state = c_rarg1;
5105       Register ofs   = c_rarg2;
5106       Register limit = c_rarg3;
5107       Register consts =  t2; // caller saved
5108       Register state_c = x28; // caller saved
5109       VectorRegister vindex = v2;
5110       VectorRegister vW0 = v4;
5111       VectorRegister vW1 = v6;
5112       VectorRegister vW2 = v8;
5113       VectorRegister vW3 = v10;
5114       VectorRegister vState0 = v12;
5115       VectorRegister vState1 = v14;
5116       VectorRegister vHash0  = v16;
5117       VectorRegister vHash1  = v18;
5118       VectorRegister vTmp0   = v20;
5119       VectorRegister vTmp1   = v22;
5120 
5121       Label multi_block_loop;
5122 
5123       __ enter();
5124 
5125       address constant_table = vset_sew == Assembler::e32 ? (address)round_consts_256 : (address)round_consts_512;
5126       la(consts, ExternalAddress(constant_table));
5127 
5128       // Register use in this function:
5129       //
5130       // VECTORS
5131       //  vW0 - vW3 (512/1024-bits / 4*128/256 bits / 4*4*32/65 bits), hold the message
5132       //             schedule words (Wt). They start with the message block
5133       //             content (W0 to W15), then further words in the message
5134       //             schedule generated via vsha2ms from previous Wt.
5135       //   Initially:
5136       //     vW0 = W[  3:0] = { W3,  W2,  W1,  W0}
5137       //     vW1 = W[  7:4] = { W7,  W6,  W5,  W4}
5138       //     vW2 = W[ 11:8] = {W11, W10,  W9,  W8}
5139       //     vW3 = W[15:12] = {W15, W14, W13, W12}
5140       //
5141       //  vState0 - vState1 hold the working state variables (a, b, ..., h)
5142       //    vState0 = {f[t],e[t],b[t],a[t]}
5143       //    vState1 = {h[t],g[t],d[t],c[t]}
5144       //   Initially:
5145       //    vState0 = {H5i-1, H4i-1, H1i-1 , H0i-1}
5146       //    vState1 = {H7i-i, H6i-1, H3i-1 , H2i-1}
5147       //
5148       //  v0 = masks for vrgather/vmerge. Single value during the 16 rounds.
5149       //
5150       //  vTmp0 = temporary, Wt+Kt
5151       //  vTmp1 = temporary, Kt
5152       //
5153       //  vHash0/vHash1 = hold the initial values of the hash, byte-swapped.
5154       //
5155       // During most of the function the vector state is configured so that each
5156       // vector is interpreted as containing four 32/64 bits (e32/e64) elements (128/256 bits).
5157 
5158       // vsha2ch/vsha2cl uses EGW of 4*SEW.
5159       // SHA256 SEW = e32, EGW = 128-bits
5160       // SHA512 SEW = e64, EGW = 256-bits
5161       //
5162       // VLEN is required to be at least 128.
5163       // For the case of VLEN=128 and SHA512 we need LMUL=2 to work with 4*e64 (EGW = 256)
5164       //
5165       // m1: LMUL=1/2
5166       // ta: tail agnostic (don't care about those lanes)
5167       // ma: mask agnostic (don't care about those lanes)
5168       // x0 is not written, we known the number of vector elements.
5169 
5170       if (vset_sew == Assembler::e64 && MaxVectorSize == 16) { // SHA512 and VLEN = 128
5171         __ vsetivli(x0, 4, vset_sew, Assembler::m2, Assembler::ma, Assembler::ta);
5172       } else {
5173         __ vsetivli(x0, 4, vset_sew, Assembler::m1, Assembler::ma, Assembler::ta);
5174       }
5175 
5176       int64_t indexes = vset_sew == Assembler::e32 ? 0x00041014ul : 0x00082028ul;
5177       __ li(t0, indexes);
5178       __ vmv_v_x(vindex, t0);
5179 
5180       // Step-over a,b, so we are pointing to c.
5181       // const_add is equal to 4x state variable, div by 2 is thus 2, a,b
5182       __ addi(state_c, state, const_add/2);
5183 
5184       // Use index-load to get {f,e,b,a},{h,g,d,c}
5185       __ vluxei8_v(vState0, state, vindex);
5186       __ vluxei8_v(vState1, state_c, vindex);
5187 
5188       __ bind(multi_block_loop);
5189 
5190       // Capture the initial H values in vHash0 and vHash1 to allow for computing
5191       // the resulting H', since H' = H+{a',b',c',...,h'}.
5192       __ vmv_v_v(vHash0, vState0);
5193       __ vmv_v_v(vHash1, vState1);
5194 
5195       // Load the 512/1024-bits of the message block in vW0-vW3 and perform
5196       // an endian swap on each 4/8 bytes element.
5197       //
5198       // If Zvkb is not implemented one can use vrgather
5199       // with an index sequence to byte-swap.
5200       //  sequence = [3 2 1 0   7 6 5 4  11 10 9 8   15 14 13 12]
5201       //   <https://oeis.org/A004444> gives us "N ^ 3" as a nice formula to generate
5202       //  this sequence. 'vid' gives us the N.
5203       __ vleXX_v(vset_sew, vW0, buf);
5204       __ vrev8_v(vW0, vW0);
5205       __ addi(buf, buf, const_add);
5206       __ vleXX_v(vset_sew, vW1, buf);
5207       __ vrev8_v(vW1, vW1);
5208       __ addi(buf, buf, const_add);
5209       __ vleXX_v(vset_sew, vW2, buf);
5210       __ vrev8_v(vW2, vW2);
5211       __ addi(buf, buf, const_add);
5212       __ vleXX_v(vset_sew, vW3, buf);
5213       __ vrev8_v(vW3, vW3);
5214       __ addi(buf, buf, const_add);
5215 
5216       // Set v0 up for the vmerge that replaces the first word (idx==0)
5217       __ vid_v(v0);
5218       __ vmseq_vi(v0, v0, 0x0);  // v0.mask[i] = (i == 0 ? 1 : 0)
5219 
5220       VectorRegister rotation_regs[] = {vW0, vW1, vW2, vW3};
5221       int rot_pos = 0;
5222       // Quad-round #0 (+0, vW0->vW1->vW2->vW3) ... #11 (+3, vW3->vW0->vW1->vW2)
5223       const int qr_end = vset_sew == Assembler::e32 ? 12 : 16;
5224       for (int i = 0; i < qr_end; i++) {
5225         sha2_quad_round(vset_sew,
5226                    rotation_regs[(rot_pos + 0) & 0x3],
5227                    rotation_regs[(rot_pos + 1) & 0x3],
5228                    rotation_regs[(rot_pos + 2) & 0x3],
5229                    rotation_regs[(rot_pos + 3) & 0x3],
5230                    consts,
5231                    vTmp1, vTmp0, vState0, vState1);
5232         ++rot_pos;
5233       }
5234       // Quad-round #12 (+0, vW0->vW1->vW2->vW3) ... #15 (+3, vW3->vW0->vW1->vW2)
5235       // Note that we stop generating new message schedule words (Wt, vW0-13)
5236       // as we already generated all the words we end up consuming (i.e., W[63:60]).
5237       const int qr_c_end = qr_end + 4;
5238       for (int i = qr_end; i < qr_c_end; i++) {
5239         sha2_quad_round(vset_sew,
5240                    rotation_regs[(rot_pos + 0) & 0x3],
5241                    rotation_regs[(rot_pos + 1) & 0x3],
5242                    rotation_regs[(rot_pos + 2) & 0x3],
5243                    rotation_regs[(rot_pos + 3) & 0x3],
5244                    consts,
5245                    vTmp1, vTmp0, vState0, vState1, false, i < (qr_c_end-1));
5246         ++rot_pos;
5247       }
5248 
5249       //--------------------------------------------------------------------------------
5250       // Compute the updated hash value H'
5251       //   H' = H + {h',g',...,b',a'}
5252       //      = {h,g,...,b,a} + {h',g',...,b',a'}
5253       //      = {h+h',g+g',...,b+b',a+a'}
5254 
5255       // H' = H+{a',b',c',...,h'}
5256       __ vadd_vv(vState0, vHash0, vState0);
5257       __ vadd_vv(vState1, vHash1, vState1);
5258 
5259       if (multi_block) {
5260         int total_adds = vset_sew == Assembler::e32 ? 240 : 608;
5261         __ subi(consts, consts, total_adds);
5262         __ addi(ofs, ofs, vset_sew == Assembler::e32 ? 64 : 128);
5263         __ ble(ofs, limit, multi_block_loop);
5264         __ mv(c_rarg0, ofs); // return ofs
5265       }
5266 
5267       // Store H[0..8] = {a,b,c,d,e,f,g,h} from
5268       //  vState0 = {f,e,b,a}
5269       //  vState1 = {h,g,d,c}
5270       __ vsuxei8_v(vState0, state,   vindex);
5271       __ vsuxei8_v(vState1, state_c, vindex);
5272 
5273       __ leave();
5274       __ ret();
5275 
5276       return start;
5277     }
5278   };
5279 
5280 #undef __
5281 #define __ _masm->
5282 
5283   // Set of L registers that correspond to a contiguous memory area.
5284   // Each 64-bit register typically corresponds to 2 32-bit integers.
5285   template <uint L>
5286   class RegCache {
5287   private:
5288     MacroAssembler *_masm;
5289     Register _regs[L];
5290 
5291   public:
5292     RegCache(MacroAssembler *masm, RegSet rs): _masm(masm) {
5293       assert(rs.size() == L, "%u registers are used to cache %u 4-byte data", rs.size(), 2 * L);
5294       auto it = rs.begin();
5295       for (auto &r: _regs) {
5296         r = *it;
5297         ++it;
5298       }
5299     }
5300 
5301     // generate load for the i'th register
5302     void gen_load(uint i, Register base) {
5303       assert(i < L, "invalid i: %u", i);
5304       __ ld(_regs[i], Address(base, 8 * i));
5305     }
5306 
5307     // add i'th 32-bit integer to dest
5308     void add_u32(const Register dest, uint i, const Register rtmp = t0) {
5309       assert(i < 2 * L, "invalid i: %u", i);
5310 
5311       if (is_even(i)) {
5312         // Use the bottom 32 bits. No need to mask off the top 32 bits
5313         // as addw will do the right thing.
5314         __ addw(dest, dest, _regs[i / 2]);
5315       } else {
5316         // Use the top 32 bits by right-shifting them.
5317         __ srli(rtmp, _regs[i / 2], 32);
5318         __ addw(dest, dest, rtmp);
5319       }
5320     }
5321   };
5322 
5323   typedef RegCache<8> BufRegCache;
5324 
5325   // a += value + x + ac;
5326   // a = Integer.rotateLeft(a, s) + b;
5327   void m5_FF_GG_HH_II_epilogue(BufRegCache& reg_cache,
5328                                Register a, Register b, Register c, Register d,
5329                                int k, int s, int t,
5330                                Register value) {
5331     // a += ac
5332     __ addw(a, a, t, t1);
5333 
5334     // a += x;
5335     reg_cache.add_u32(a, k);
5336     // a += value;
5337     __ addw(a, a, value);
5338 
5339     // a = Integer.rotateLeft(a, s) + b;
5340     __ rolw(a, a, s);
5341     __ addw(a, a, b);
5342   }
5343 
5344   // a += ((b & c) | ((~b) & d)) + x + ac;
5345   // a = Integer.rotateLeft(a, s) + b;
5346   void md5_FF(BufRegCache& reg_cache,
5347               Register a, Register b, Register c, Register d,
5348               int k, int s, int t,
5349               Register rtmp1, Register rtmp2) {
5350     // rtmp1 = b & c
5351     __ andr(rtmp1, b, c);
5352 
5353     // rtmp2 = (~b) & d
5354     __ andn(rtmp2, d, b);
5355 
5356     // rtmp1 = (b & c) | ((~b) & d)
5357     __ orr(rtmp1, rtmp1, rtmp2);
5358 
5359     m5_FF_GG_HH_II_epilogue(reg_cache, a, b, c, d, k, s, t, rtmp1);
5360   }
5361 
5362   // a += ((b & d) | (c & (~d))) + x + ac;
5363   // a = Integer.rotateLeft(a, s) + b;
5364   void md5_GG(BufRegCache& reg_cache,
5365               Register a, Register b, Register c, Register d,
5366               int k, int s, int t,
5367               Register rtmp1, Register rtmp2) {
5368     // rtmp1 = b & d
5369     __ andr(rtmp1, b, d);
5370 
5371     // rtmp2 = c & (~d)
5372     __ andn(rtmp2, c, d);
5373 
5374     // rtmp1 = (b & d) | (c & (~d))
5375     __ orr(rtmp1, rtmp1, rtmp2);
5376 
5377     m5_FF_GG_HH_II_epilogue(reg_cache, a, b, c, d, k, s, t, rtmp1);
5378   }
5379 
5380   // a += ((b ^ c) ^ d) + x + ac;
5381   // a = Integer.rotateLeft(a, s) + b;
5382   void md5_HH(BufRegCache& reg_cache,
5383               Register a, Register b, Register c, Register d,
5384               int k, int s, int t,
5385               Register rtmp1, Register rtmp2) {
5386     // rtmp1 = (b ^ c) ^ d
5387     __ xorr(rtmp2, b, c);
5388     __ xorr(rtmp1, rtmp2, d);
5389 
5390     m5_FF_GG_HH_II_epilogue(reg_cache, a, b, c, d, k, s, t, rtmp1);
5391   }
5392 
5393   // a += (c ^ (b | (~d))) + x + ac;
5394   // a = Integer.rotateLeft(a, s) + b;
5395   void md5_II(BufRegCache& reg_cache,
5396               Register a, Register b, Register c, Register d,
5397               int k, int s, int t,
5398               Register rtmp1, Register rtmp2) {
5399     // rtmp1 = c ^ (b | (~d))
5400     __ orn(rtmp2, b, d);
5401     __ xorr(rtmp1, c, rtmp2);
5402 
5403     m5_FF_GG_HH_II_epilogue(reg_cache, a, b, c, d, k, s, t, rtmp1);
5404   }
5405 
5406   // Arguments:
5407   //
5408   // Inputs:
5409   //   c_rarg0   - byte[]  source+offset
5410   //   c_rarg1   - int[]   SHA.state
5411   //   c_rarg2   - int     offset  (multi_block == True)
5412   //   c_rarg3   - int     limit   (multi_block == True)
5413   //
5414   // Registers:
5415   //    x0   zero  (zero)
5416   //    x1     ra  (return address)
5417   //    x2     sp  (stack pointer)
5418   //    x3     gp  (global pointer)
5419   //    x4     tp  (thread pointer)
5420   //    x5     t0  (tmp register)
5421   //    x6     t1  (tmp register)
5422   //    x7     t2  state0
5423   //    x8  f0/s0  (frame pointer)
5424   //    x9     s1
5425   //   x10     a0  rtmp1 / c_rarg0
5426   //   x11     a1  rtmp2 / c_rarg1
5427   //   x12     a2  a     / c_rarg2
5428   //   x13     a3  b     / c_rarg3
5429   //   x14     a4  c
5430   //   x15     a5  d
5431   //   x16     a6  buf
5432   //   x17     a7  state
5433   //   x18     s2  ofs     [saved-reg]  (multi_block == True)
5434   //   x19     s3  limit   [saved-reg]  (multi_block == True)
5435   //   x20     s4  state1  [saved-reg]
5436   //   x21     s5  state2  [saved-reg]
5437   //   x22     s6  state3  [saved-reg]
5438   //   x23     s7
5439   //   x24     s8  buf0    [saved-reg]
5440   //   x25     s9  buf1    [saved-reg]
5441   //   x26    s10  buf2    [saved-reg]
5442   //   x27    s11  buf3    [saved-reg]
5443   //   x28     t3  buf4
5444   //   x29     t4  buf5
5445   //   x30     t5  buf6
5446   //   x31     t6  buf7
5447   address generate_md5_implCompress(StubId stub_id) {
5448     __ align(CodeEntryAlignment);
5449     bool multi_block;
5450     switch (stub_id) {
5451     case StubId::stubgen_md5_implCompress_id:
5452       multi_block = false;
5453       break;
5454     case StubId::stubgen_md5_implCompressMB_id:
5455       multi_block = true;
5456       break;
5457     default:
5458       ShouldNotReachHere();
5459     };
5460     StubCodeMark mark(this, stub_id);
5461     address start = __ pc();
5462 
5463     // rotation constants
5464     const int S11 = 7;
5465     const int S12 = 12;
5466     const int S13 = 17;
5467     const int S14 = 22;
5468     const int S21 = 5;
5469     const int S22 = 9;
5470     const int S23 = 14;
5471     const int S24 = 20;
5472     const int S31 = 4;
5473     const int S32 = 11;
5474     const int S33 = 16;
5475     const int S34 = 23;
5476     const int S41 = 6;
5477     const int S42 = 10;
5478     const int S43 = 15;
5479     const int S44 = 21;
5480 
5481     const int64_t mask32 = 0xffffffff;
5482 
5483     Register buf_arg   = c_rarg0; // a0
5484     Register state_arg = c_rarg1; // a1
5485     Register ofs_arg   = c_rarg2; // a2
5486     Register limit_arg = c_rarg3; // a3
5487 
5488     // we'll copy the args to these registers to free up a0-a3
5489     // to use for other values manipulated by instructions
5490     // that can be compressed
5491     Register buf       = x16; // a6
5492     Register state     = x17; // a7
5493     Register ofs       = x18; // s2
5494     Register limit     = x19; // s3
5495 
5496     // using x12->15 to allow compressed instructions
5497     Register a         = x12; // a2
5498     Register b         = x13; // a3
5499     Register c         = x14; // a4
5500     Register d         = x15; // a5
5501 
5502     Register state0    =  x7; // t2
5503     Register state1    = x20; // s4
5504     Register state2    = x21; // s5
5505     Register state3    = x22; // s6
5506 
5507     // using x10->x11 to allow compressed instructions
5508     Register rtmp1     = x10; // a0
5509     Register rtmp2     = x11; // a1
5510 
5511     RegSet reg_cache_saved_regs = RegSet::of(x24, x25, x26, x27); // s8, s9, s10, s11
5512     RegSet reg_cache_regs;
5513     reg_cache_regs += reg_cache_saved_regs;
5514     reg_cache_regs += RegSet::of(t3, t4, t5, t6);
5515     BufRegCache reg_cache(_masm, reg_cache_regs);
5516 
5517     RegSet saved_regs;
5518     if (multi_block) {
5519       saved_regs += RegSet::of(ofs, limit);
5520     }
5521     saved_regs += RegSet::of(state1, state2, state3);
5522     saved_regs += reg_cache_saved_regs;
5523 
5524     __ push_reg(saved_regs, sp);
5525 
5526     __ mv(buf, buf_arg);
5527     __ mv(state, state_arg);
5528     if (multi_block) {
5529       __ mv(ofs, ofs_arg);
5530       __ mv(limit, limit_arg);
5531     }
5532 
5533     // to minimize the number of memory operations:
5534     // read the 4 state 4-byte values in pairs, with a single ld,
5535     // and split them into 2 registers.
5536     //
5537     // And, as the core algorithm of md5 works on 32-bits words, so
5538     // in the following code, it does not care about the content of
5539     // higher 32-bits in state[x]. Based on this observation,
5540     // we can apply further optimization, which is to just ignore the
5541     // higher 32-bits in state0/state2, rather than set the higher
5542     // 32-bits of state0/state2 to zero explicitly with extra instructions.
5543     __ ld(state0, Address(state));
5544     __ srli(state1, state0, 32);
5545     __ ld(state2, Address(state, 8));
5546     __ srli(state3, state2, 32);
5547 
5548     Label md5_loop;
5549     __ BIND(md5_loop);
5550 
5551     __ mv(a, state0);
5552     __ mv(b, state1);
5553     __ mv(c, state2);
5554     __ mv(d, state3);
5555 
5556     // Round 1
5557     reg_cache.gen_load(0, buf);
5558     md5_FF(reg_cache, a, b, c, d,  0, S11, 0xd76aa478, rtmp1, rtmp2);
5559     md5_FF(reg_cache, d, a, b, c,  1, S12, 0xe8c7b756, rtmp1, rtmp2);
5560     reg_cache.gen_load(1, buf);
5561     md5_FF(reg_cache, c, d, a, b,  2, S13, 0x242070db, rtmp1, rtmp2);
5562     md5_FF(reg_cache, b, c, d, a,  3, S14, 0xc1bdceee, rtmp1, rtmp2);
5563     reg_cache.gen_load(2, buf);
5564     md5_FF(reg_cache, a, b, c, d,  4, S11, 0xf57c0faf, rtmp1, rtmp2);
5565     md5_FF(reg_cache, d, a, b, c,  5, S12, 0x4787c62a, rtmp1, rtmp2);
5566     reg_cache.gen_load(3, buf);
5567     md5_FF(reg_cache, c, d, a, b,  6, S13, 0xa8304613, rtmp1, rtmp2);
5568     md5_FF(reg_cache, b, c, d, a,  7, S14, 0xfd469501, rtmp1, rtmp2);
5569     reg_cache.gen_load(4, buf);
5570     md5_FF(reg_cache, a, b, c, d,  8, S11, 0x698098d8, rtmp1, rtmp2);
5571     md5_FF(reg_cache, d, a, b, c,  9, S12, 0x8b44f7af, rtmp1, rtmp2);
5572     reg_cache.gen_load(5, buf);
5573     md5_FF(reg_cache, c, d, a, b, 10, S13, 0xffff5bb1, rtmp1, rtmp2);
5574     md5_FF(reg_cache, b, c, d, a, 11, S14, 0x895cd7be, rtmp1, rtmp2);
5575     reg_cache.gen_load(6, buf);
5576     md5_FF(reg_cache, a, b, c, d, 12, S11, 0x6b901122, rtmp1, rtmp2);
5577     md5_FF(reg_cache, d, a, b, c, 13, S12, 0xfd987193, rtmp1, rtmp2);
5578     reg_cache.gen_load(7, buf);
5579     md5_FF(reg_cache, c, d, a, b, 14, S13, 0xa679438e, rtmp1, rtmp2);
5580     md5_FF(reg_cache, b, c, d, a, 15, S14, 0x49b40821, rtmp1, rtmp2);
5581 
5582     // Round 2
5583     md5_GG(reg_cache, a, b, c, d,  1, S21, 0xf61e2562, rtmp1, rtmp2);
5584     md5_GG(reg_cache, d, a, b, c,  6, S22, 0xc040b340, rtmp1, rtmp2);
5585     md5_GG(reg_cache, c, d, a, b, 11, S23, 0x265e5a51, rtmp1, rtmp2);
5586     md5_GG(reg_cache, b, c, d, a,  0, S24, 0xe9b6c7aa, rtmp1, rtmp2);
5587     md5_GG(reg_cache, a, b, c, d,  5, S21, 0xd62f105d, rtmp1, rtmp2);
5588     md5_GG(reg_cache, d, a, b, c, 10, S22, 0x02441453, rtmp1, rtmp2);
5589     md5_GG(reg_cache, c, d, a, b, 15, S23, 0xd8a1e681, rtmp1, rtmp2);
5590     md5_GG(reg_cache, b, c, d, a,  4, S24, 0xe7d3fbc8, rtmp1, rtmp2);
5591     md5_GG(reg_cache, a, b, c, d,  9, S21, 0x21e1cde6, rtmp1, rtmp2);
5592     md5_GG(reg_cache, d, a, b, c, 14, S22, 0xc33707d6, rtmp1, rtmp2);
5593     md5_GG(reg_cache, c, d, a, b,  3, S23, 0xf4d50d87, rtmp1, rtmp2);
5594     md5_GG(reg_cache, b, c, d, a,  8, S24, 0x455a14ed, rtmp1, rtmp2);
5595     md5_GG(reg_cache, a, b, c, d, 13, S21, 0xa9e3e905, rtmp1, rtmp2);
5596     md5_GG(reg_cache, d, a, b, c,  2, S22, 0xfcefa3f8, rtmp1, rtmp2);
5597     md5_GG(reg_cache, c, d, a, b,  7, S23, 0x676f02d9, rtmp1, rtmp2);
5598     md5_GG(reg_cache, b, c, d, a, 12, S24, 0x8d2a4c8a, rtmp1, rtmp2);
5599 
5600     // Round 3
5601     md5_HH(reg_cache, a, b, c, d,  5, S31, 0xfffa3942, rtmp1, rtmp2);
5602     md5_HH(reg_cache, d, a, b, c,  8, S32, 0x8771f681, rtmp1, rtmp2);
5603     md5_HH(reg_cache, c, d, a, b, 11, S33, 0x6d9d6122, rtmp1, rtmp2);
5604     md5_HH(reg_cache, b, c, d, a, 14, S34, 0xfde5380c, rtmp1, rtmp2);
5605     md5_HH(reg_cache, a, b, c, d,  1, S31, 0xa4beea44, rtmp1, rtmp2);
5606     md5_HH(reg_cache, d, a, b, c,  4, S32, 0x4bdecfa9, rtmp1, rtmp2);
5607     md5_HH(reg_cache, c, d, a, b,  7, S33, 0xf6bb4b60, rtmp1, rtmp2);
5608     md5_HH(reg_cache, b, c, d, a, 10, S34, 0xbebfbc70, rtmp1, rtmp2);
5609     md5_HH(reg_cache, a, b, c, d, 13, S31, 0x289b7ec6, rtmp1, rtmp2);
5610     md5_HH(reg_cache, d, a, b, c,  0, S32, 0xeaa127fa, rtmp1, rtmp2);
5611     md5_HH(reg_cache, c, d, a, b,  3, S33, 0xd4ef3085, rtmp1, rtmp2);
5612     md5_HH(reg_cache, b, c, d, a,  6, S34, 0x04881d05, rtmp1, rtmp2);
5613     md5_HH(reg_cache, a, b, c, d,  9, S31, 0xd9d4d039, rtmp1, rtmp2);
5614     md5_HH(reg_cache, d, a, b, c, 12, S32, 0xe6db99e5, rtmp1, rtmp2);
5615     md5_HH(reg_cache, c, d, a, b, 15, S33, 0x1fa27cf8, rtmp1, rtmp2);
5616     md5_HH(reg_cache, b, c, d, a,  2, S34, 0xc4ac5665, rtmp1, rtmp2);
5617 
5618     // Round 4
5619     md5_II(reg_cache, a, b, c, d,  0, S41, 0xf4292244, rtmp1, rtmp2);
5620     md5_II(reg_cache, d, a, b, c,  7, S42, 0x432aff97, rtmp1, rtmp2);
5621     md5_II(reg_cache, c, d, a, b, 14, S43, 0xab9423a7, rtmp1, rtmp2);
5622     md5_II(reg_cache, b, c, d, a,  5, S44, 0xfc93a039, rtmp1, rtmp2);
5623     md5_II(reg_cache, a, b, c, d, 12, S41, 0x655b59c3, rtmp1, rtmp2);
5624     md5_II(reg_cache, d, a, b, c,  3, S42, 0x8f0ccc92, rtmp1, rtmp2);
5625     md5_II(reg_cache, c, d, a, b, 10, S43, 0xffeff47d, rtmp1, rtmp2);
5626     md5_II(reg_cache, b, c, d, a,  1, S44, 0x85845dd1, rtmp1, rtmp2);
5627     md5_II(reg_cache, a, b, c, d,  8, S41, 0x6fa87e4f, rtmp1, rtmp2);
5628     md5_II(reg_cache, d, a, b, c, 15, S42, 0xfe2ce6e0, rtmp1, rtmp2);
5629     md5_II(reg_cache, c, d, a, b,  6, S43, 0xa3014314, rtmp1, rtmp2);
5630     md5_II(reg_cache, b, c, d, a, 13, S44, 0x4e0811a1, rtmp1, rtmp2);
5631     md5_II(reg_cache, a, b, c, d,  4, S41, 0xf7537e82, rtmp1, rtmp2);
5632     md5_II(reg_cache, d, a, b, c, 11, S42, 0xbd3af235, rtmp1, rtmp2);
5633     md5_II(reg_cache, c, d, a, b,  2, S43, 0x2ad7d2bb, rtmp1, rtmp2);
5634     md5_II(reg_cache, b, c, d, a,  9, S44, 0xeb86d391, rtmp1, rtmp2);
5635 
5636     __ addw(state0, state0, a);
5637     __ addw(state1, state1, b);
5638     __ addw(state2, state2, c);
5639     __ addw(state3, state3, d);
5640 
5641     if (multi_block) {
5642       __ addi(buf, buf, 64);
5643       __ addi(ofs, ofs, 64);
5644       // if (ofs <= limit) goto m5_loop
5645       __ bge(limit, ofs, md5_loop);
5646       __ mv(c_rarg0, ofs); // return ofs
5647     }
5648 
5649     // to minimize the number of memory operations:
5650     // write back the 4 state 4-byte values in pairs, with a single sd
5651     __ mv(t0, mask32);
5652     __ andr(state0, state0, t0);
5653     __ slli(state1, state1, 32);
5654     __ orr(state0, state0, state1);
5655     __ sd(state0, Address(state));
5656     __ andr(state2, state2, t0);
5657     __ slli(state3, state3, 32);
5658     __ orr(state2, state2, state3);
5659     __ sd(state2, Address(state, 8));
5660 
5661     __ pop_reg(saved_regs, sp);
5662     __ ret();
5663 
5664     return (address) start;
5665   }
5666 
5667   /**
5668    * Perform the quarter round calculations on values contained within four vector registers.
5669    *
5670    * @param aVec the SIMD register containing only the "a" values
5671    * @param bVec the SIMD register containing only the "b" values
5672    * @param cVec the SIMD register containing only the "c" values
5673    * @param dVec the SIMD register containing only the "d" values
5674    * @param tmp_vr temporary vector register holds intermedia values.
5675    */
5676   void chacha20_quarter_round(VectorRegister aVec, VectorRegister bVec,
5677                           VectorRegister cVec, VectorRegister dVec, VectorRegister tmp_vr) {
5678     // a += b, d ^= a, d <<<= 16
5679     __ vadd_vv(aVec, aVec, bVec);
5680     __ vxor_vv(dVec, dVec, aVec);
5681     __ vrole32_vi(dVec, 16, tmp_vr);
5682 
5683     // c += d, b ^= c, b <<<= 12
5684     __ vadd_vv(cVec, cVec, dVec);
5685     __ vxor_vv(bVec, bVec, cVec);
5686     __ vrole32_vi(bVec, 12, tmp_vr);
5687 
5688     // a += b, d ^= a, d <<<= 8
5689     __ vadd_vv(aVec, aVec, bVec);
5690     __ vxor_vv(dVec, dVec, aVec);
5691     __ vrole32_vi(dVec, 8, tmp_vr);
5692 
5693     // c += d, b ^= c, b <<<= 7
5694     __ vadd_vv(cVec, cVec, dVec);
5695     __ vxor_vv(bVec, bVec, cVec);
5696     __ vrole32_vi(bVec, 7, tmp_vr);
5697   }
5698 
5699   /**
5700    * int com.sun.crypto.provider.ChaCha20Cipher.implChaCha20Block(int[] initState, byte[] result)
5701    *
5702    *  Input arguments:
5703    *  c_rarg0   - state, the starting state
5704    *  c_rarg1   - key_stream, the array that will hold the result of the ChaCha20 block function
5705    *
5706    *  Implementation Note:
5707    *   Parallelization is achieved by loading individual state elements into vectors for N blocks.
5708    *   N depends on single vector register length.
5709    */
5710   address generate_chacha20Block() {
5711     Label L_Rounds;
5712 
5713     __ align(CodeEntryAlignment);
5714     StubId stub_id = StubId::stubgen_chacha20Block_id;
5715     StubCodeMark mark(this, stub_id);
5716     address start = __ pc();
5717     __ enter();
5718 
5719     const int states_len = 16;
5720     const int step = 4;
5721     const Register state = c_rarg0;
5722     const Register key_stream = c_rarg1;
5723     const Register tmp_addr = t0;
5724     const Register length = t1;
5725 
5726     // Organize vector registers in an array that facilitates
5727     // putting repetitive opcodes into loop structures below.
5728     const VectorRegister work_vrs[16] = {
5729       v0, v1, v2,  v3,  v4,  v5,  v6,  v7,
5730       v8, v9, v10, v11, v12, v13, v14, v15
5731     };
5732     const VectorRegister tmp_vr = v16;
5733     const VectorRegister counter_vr = v17;
5734 
5735     {
5736       // Put 16 here, as com.sun.crypto.providerChaCha20Cipher.KS_MAX_LEN is 1024
5737       // in java level.
5738       __ vsetivli(length, 16, Assembler::e32, Assembler::m1);
5739     }
5740 
5741     // Load from source state.
5742     // Every element in source state is duplicated to all elements in the corresponding vector.
5743     __ mv(tmp_addr, state);
5744     for (int i = 0; i < states_len; i += 1) {
5745       __ vlse32_v(work_vrs[i], tmp_addr, zr);
5746       __ addi(tmp_addr, tmp_addr, step);
5747     }
5748     // Adjust counter for every individual block.
5749     __ vid_v(counter_vr);
5750     __ vadd_vv(work_vrs[12], work_vrs[12], counter_vr);
5751 
5752     // Perform 10 iterations of the 8 quarter round set
5753     {
5754       const Register loop = t2; // share t2 with other non-overlapping usages.
5755       __ mv(loop, 10);
5756       __ BIND(L_Rounds);
5757 
5758       chacha20_quarter_round(work_vrs[0], work_vrs[4], work_vrs[8],  work_vrs[12], tmp_vr);
5759       chacha20_quarter_round(work_vrs[1], work_vrs[5], work_vrs[9],  work_vrs[13], tmp_vr);
5760       chacha20_quarter_round(work_vrs[2], work_vrs[6], work_vrs[10], work_vrs[14], tmp_vr);
5761       chacha20_quarter_round(work_vrs[3], work_vrs[7], work_vrs[11], work_vrs[15], tmp_vr);
5762 
5763       chacha20_quarter_round(work_vrs[0], work_vrs[5], work_vrs[10], work_vrs[15], tmp_vr);
5764       chacha20_quarter_round(work_vrs[1], work_vrs[6], work_vrs[11], work_vrs[12], tmp_vr);
5765       chacha20_quarter_round(work_vrs[2], work_vrs[7], work_vrs[8],  work_vrs[13], tmp_vr);
5766       chacha20_quarter_round(work_vrs[3], work_vrs[4], work_vrs[9],  work_vrs[14], tmp_vr);
5767 
5768       __ subi(loop, loop, 1);
5769       __ bnez(loop, L_Rounds);
5770     }
5771 
5772     // Add the original state into the end working state.
5773     // We do this by first duplicating every element in source state array to the corresponding
5774     // vector, then adding it to the post-loop working state.
5775     __ mv(tmp_addr, state);
5776     for (int i = 0; i < states_len; i += 1) {
5777       __ vlse32_v(tmp_vr, tmp_addr, zr);
5778       __ addi(tmp_addr, tmp_addr, step);
5779       __ vadd_vv(work_vrs[i], work_vrs[i], tmp_vr);
5780     }
5781     // Add the counter overlay onto work_vrs[12] at the end.
5782     __ vadd_vv(work_vrs[12], work_vrs[12], counter_vr);
5783 
5784     // Store result to key stream.
5785     {
5786       const Register stride = t2; // share t2 with other non-overlapping usages.
5787       // Every block occupies 64 bytes, so we use 64 as stride of the vector store.
5788       __ mv(stride, 64);
5789       for (int i = 0; i < states_len; i += 1) {
5790         __ vsse32_v(work_vrs[i], key_stream, stride);
5791         __ addi(key_stream, key_stream, step);
5792       }
5793     }
5794 
5795     // Return length of output key_stream
5796     __ slli(c_rarg0, length, 6);
5797 
5798     __ leave();
5799     __ ret();
5800 
5801     return (address) start;
5802   }
5803 
5804 
5805   // ------------------------ SHA-1 intrinsic ------------------------
5806 
5807   // K't =
5808   //    5a827999, 0  <= t <= 19
5809   //    6ed9eba1, 20 <= t <= 39
5810   //    8f1bbcdc, 40 <= t <= 59
5811   //    ca62c1d6, 60 <= t <= 79
5812   void sha1_prepare_k(Register cur_k, int round) {
5813     assert(round >= 0 && round < 80, "must be");
5814 
5815     static const int64_t ks[] = {0x5a827999, 0x6ed9eba1, 0x8f1bbcdc, 0xca62c1d6};
5816     if ((round % 20) == 0) {
5817       __ mv(cur_k, ks[round/20]);
5818     }
5819   }
5820 
5821   // W't =
5822   //    M't,                                      0 <=  t <= 15
5823   //    ROTL'1(W't-3 ^ W't-8 ^ W't-14 ^ W't-16),  16 <= t <= 79
5824   void sha1_prepare_w(Register cur_w, Register ws[], Register buf, int round) {
5825     assert(round >= 0 && round < 80, "must be");
5826 
5827     if (round < 16) {
5828       // in the first 16 rounds, in ws[], every register contains 2 W't, e.g.
5829       //   in ws[0], high part contains W't-0, low part contains W't-1,
5830       //   in ws[1], high part contains W't-2, low part contains W't-3,
5831       //   ...
5832       //   in ws[7], high part contains W't-14, low part contains W't-15.
5833 
5834       if ((round % 2) == 0) {
5835         __ ld(ws[round/2], Address(buf, (round/2) * 8));
5836         // reverse bytes, as SHA-1 is defined in big-endian.
5837         __ revb(ws[round/2], ws[round/2]);
5838         __ srli(cur_w, ws[round/2], 32);
5839       } else {
5840         __ mv(cur_w, ws[round/2]);
5841       }
5842 
5843       return;
5844     }
5845 
5846     if ((round % 2) == 0) {
5847       int idx = 16;
5848       // W't = ROTL'1(W't-3 ^ W't-8 ^ W't-14 ^ W't-16),  16 <= t <= 79
5849       __ srli(t1, ws[(idx-8)/2], 32);
5850       __ xorr(t0, ws[(idx-3)/2], t1);
5851 
5852       __ srli(t1, ws[(idx-14)/2], 32);
5853       __ srli(cur_w, ws[(idx-16)/2], 32);
5854       __ xorr(cur_w, cur_w, t1);
5855 
5856       __ xorr(cur_w, cur_w, t0);
5857       __ rolw(cur_w, cur_w, 1, t0);
5858 
5859       // copy the cur_w value to ws[8].
5860       // now, valid w't values are at:
5861       //  w0:       ws[0]'s lower 32 bits
5862       //  w1 ~ w14: ws[1] ~ ws[7]
5863       //  w15:      ws[8]'s higher 32 bits
5864       __ slli(ws[idx/2], cur_w, 32);
5865 
5866       return;
5867     }
5868 
5869     int idx = 17;
5870     // W't = ROTL'1(W't-3 ^ W't-8 ^ W't-14 ^ W't-16),  16 <= t <= 79
5871     __ srli(t1, ws[(idx-3)/2], 32);
5872     __ xorr(t0, t1, ws[(idx-8)/2]);
5873 
5874     __ xorr(cur_w, ws[(idx-16)/2], ws[(idx-14)/2]);
5875 
5876     __ xorr(cur_w, cur_w, t0);
5877     __ rolw(cur_w, cur_w, 1, t0);
5878 
5879     // copy the cur_w value to ws[8]
5880     __ zext(cur_w, cur_w, 32);
5881     __ orr(ws[idx/2], ws[idx/2], cur_w);
5882 
5883     // shift the w't registers, so they start from ws[0] again.
5884     // now, valid w't values are at:
5885     //  w0 ~ w15: ws[0] ~ ws[7]
5886     Register ws_0 = ws[0];
5887     for (int i = 0; i < 16/2; i++) {
5888       ws[i] = ws[i+1];
5889     }
5890     ws[8] = ws_0;
5891   }
5892 
5893   // f't(x, y, z) =
5894   //    Ch(x, y, z)     = (x & y) ^ (~x & z)            , 0  <= t <= 19
5895   //    Parity(x, y, z) = x ^ y ^ z                     , 20 <= t <= 39
5896   //    Maj(x, y, z)    = (x & y) ^ (x & z) ^ (y & z)   , 40 <= t <= 59
5897   //    Parity(x, y, z) = x ^ y ^ z                     , 60 <= t <= 79
5898   void sha1_f(Register dst, Register x, Register y, Register z, int round) {
5899     assert(round >= 0 && round < 80, "must be");
5900     assert_different_registers(dst, x, y, z, t0, t1);
5901 
5902     if (round < 20) {
5903       // (x & y) ^ (~x & z)
5904       __ andr(t0, x, y);
5905       __ andn(dst, z, x);
5906       __ xorr(dst, dst, t0);
5907     } else if (round >= 40 && round < 60) {
5908       // (x & y) ^ (x & z) ^ (y & z)
5909       __ andr(t0, x, y);
5910       __ andr(t1, x, z);
5911       __ andr(dst, y, z);
5912       __ xorr(dst, dst, t0);
5913       __ xorr(dst, dst, t1);
5914     } else {
5915       // x ^ y ^ z
5916       __ xorr(dst, x, y);
5917       __ xorr(dst, dst, z);
5918     }
5919   }
5920 
5921   // T = ROTL'5(a) + f't(b, c, d) + e + K't + W't
5922   // e = d
5923   // d = c
5924   // c = ROTL'30(b)
5925   // b = a
5926   // a = T
5927   void sha1_process_round(Register a, Register b, Register c, Register d, Register e,
5928                           Register cur_k, Register cur_w, Register tmp, int round) {
5929     assert(round >= 0 && round < 80, "must be");
5930     assert_different_registers(a, b, c, d, e, cur_w, cur_k, tmp, t0);
5931 
5932     // T = ROTL'5(a) + f't(b, c, d) + e + K't + W't
5933 
5934     // cur_w will be recalculated at the beginning of each round,
5935     // so, we can reuse it as a temp register here.
5936     Register tmp2 = cur_w;
5937 
5938     // reuse e as a temporary register, as we will mv new value into it later
5939     Register tmp3 = e;
5940     __ add(tmp2, cur_k, tmp2);
5941     __ add(tmp3, tmp3, tmp2);
5942     __ rolw(tmp2, a, 5, t0);
5943 
5944     sha1_f(tmp, b, c, d, round);
5945 
5946     __ add(tmp2, tmp2, tmp);
5947     __ add(tmp2, tmp2, tmp3);
5948 
5949     // e = d
5950     // d = c
5951     // c = ROTL'30(b)
5952     // b = a
5953     // a = T
5954     __ mv(e, d);
5955     __ mv(d, c);
5956 
5957     __ rolw(c, b, 30);
5958     __ mv(b, a);
5959     __ mv(a, tmp2);
5960   }
5961 
5962   // H(i)0 = a + H(i-1)0
5963   // H(i)1 = b + H(i-1)1
5964   // H(i)2 = c + H(i-1)2
5965   // H(i)3 = d + H(i-1)3
5966   // H(i)4 = e + H(i-1)4
5967   void sha1_calculate_im_hash(Register a, Register b, Register c, Register d, Register e,
5968                               Register prev_ab, Register prev_cd, Register prev_e) {
5969     assert_different_registers(a, b, c, d, e, prev_ab, prev_cd, prev_e);
5970 
5971     __ add(a, a, prev_ab);
5972     __ srli(prev_ab, prev_ab, 32);
5973     __ add(b, b, prev_ab);
5974 
5975     __ add(c, c, prev_cd);
5976     __ srli(prev_cd, prev_cd, 32);
5977     __ add(d, d, prev_cd);
5978 
5979     __ add(e, e, prev_e);
5980   }
5981 
5982   void sha1_preserve_prev_abcde(Register a, Register b, Register c, Register d, Register e,
5983                                 Register prev_ab, Register prev_cd, Register prev_e) {
5984     assert_different_registers(a, b, c, d, e, prev_ab, prev_cd, prev_e, t0);
5985 
5986     __ slli(t0, b, 32);
5987     __ zext(prev_ab, a, 32);
5988     __ orr(prev_ab, prev_ab, t0);
5989 
5990     __ slli(t0, d, 32);
5991     __ zext(prev_cd, c, 32);
5992     __ orr(prev_cd, prev_cd, t0);
5993 
5994     __ mv(prev_e, e);
5995   }
5996 
5997   // Intrinsic for:
5998   //   void sun.security.provider.SHA.implCompress0(byte[] buf, int ofs)
5999   //   void sun.security.provider.DigestBase.implCompressMultiBlock0(byte[] b, int ofs, int limit)
6000   //
6001   // Arguments:
6002   //
6003   // Inputs:
6004   //   c_rarg0: byte[]  src array + offset
6005   //   c_rarg1: int[]   SHA.state
6006   //   - - - - - - below are only for implCompressMultiBlock0 - - - - - -
6007   //   c_rarg2: int     offset
6008   //   c_rarg3: int     limit
6009   //
6010   // Outputs:
6011   //   - - - - - - below are only for implCompressMultiBlock0 - - - - - -
6012   //   c_rarg0: int offset, when (multi_block == true)
6013   //
6014   address generate_sha1_implCompress(StubId stub_id) {
6015       bool multi_block;
6016       switch (stub_id) {
6017       case StubId::stubgen_sha1_implCompress_id:
6018         multi_block = false;
6019         break;
6020       case StubId::stubgen_sha1_implCompressMB_id:
6021         multi_block = true;
6022         break;
6023       default:
6024         ShouldNotReachHere();
6025       };
6026     __ align(CodeEntryAlignment);
6027     StubCodeMark mark(this, stub_id);
6028 
6029     address start = __ pc();
6030     __ enter();
6031 
6032     RegSet saved_regs = RegSet::range(x18, x27);
6033     if (multi_block) {
6034       // use x9 as src below.
6035       saved_regs += RegSet::of(x9);
6036     }
6037     __ push_reg(saved_regs, sp);
6038 
6039     // c_rarg0 - c_rarg3: x10 - x13
6040     Register buf    = c_rarg0;
6041     Register state  = c_rarg1;
6042     Register offset = c_rarg2;
6043     Register limit  = c_rarg3;
6044     // use src to contain the original start point of the array.
6045     Register src    = x9;
6046 
6047     if (multi_block) {
6048       __ sub(limit, limit, offset);
6049       __ add(limit, limit, buf);
6050       __ sub(src, buf, offset);
6051     }
6052 
6053     // [args-reg]:  x14 - x17
6054     // [temp-reg]:  x28 - x31
6055     // [saved-reg]: x18 - x27
6056 
6057     // h0/1/2/3/4
6058     const Register a = x14, b = x15, c = x16, d = x17, e = x28;
6059     // w0, w1, ... w15
6060     // put two adjecent w's in one register:
6061     //    one at high word part, another at low word part
6062     // at different round (even or odd), w't value reside in different items in ws[].
6063     // w0 ~ w15, either reside in
6064     //    ws[0] ~ ws[7], where
6065     //      w0 at higher 32 bits of ws[0],
6066     //      w1 at lower 32 bits of ws[0],
6067     //      ...
6068     //      w14 at higher 32 bits of ws[7],
6069     //      w15 at lower 32 bits of ws[7].
6070     // or, reside in
6071     //    w0:       ws[0]'s lower 32 bits
6072     //    w1 ~ w14: ws[1] ~ ws[7]
6073     //    w15:      ws[8]'s higher 32 bits
6074     Register ws[9] = {x29, x30, x31, x18,
6075                       x19, x20, x21, x22,
6076                       x23}; // auxiliary register for calculating w's value
6077     // current k't's value
6078     const Register cur_k = x24;
6079     // current w't's value
6080     const Register cur_w = x25;
6081     // values of a, b, c, d, e in the previous round
6082     const Register prev_ab = x26, prev_cd = x27;
6083     const Register prev_e = offset; // reuse offset/c_rarg2
6084 
6085     // load 5 words state into a, b, c, d, e.
6086     //
6087     // To minimize the number of memory operations, we apply following
6088     // optimization: read the states (a/b/c/d) of 4-byte values in pairs,
6089     // with a single ld, and split them into 2 registers.
6090     //
6091     // And, as the core algorithm of SHA-1 works on 32-bits words, so
6092     // in the following code, it does not care about the content of
6093     // higher 32-bits in a/b/c/d/e. Based on this observation,
6094     // we can apply further optimization, which is to just ignore the
6095     // higher 32-bits in a/c/e, rather than set the higher
6096     // 32-bits of a/c/e to zero explicitly with extra instructions.
6097     __ ld(a, Address(state, 0));
6098     __ srli(b, a, 32);
6099     __ ld(c, Address(state, 8));
6100     __ srli(d, c, 32);
6101     __ lw(e, Address(state, 16));
6102 
6103     Label L_sha1_loop;
6104     if (multi_block) {
6105       __ BIND(L_sha1_loop);
6106     }
6107 
6108     sha1_preserve_prev_abcde(a, b, c, d, e, prev_ab, prev_cd, prev_e);
6109 
6110     for (int round = 0; round < 80; round++) {
6111       // prepare K't value
6112       sha1_prepare_k(cur_k, round);
6113 
6114       // prepare W't value
6115       sha1_prepare_w(cur_w, ws, buf, round);
6116 
6117       // one round process
6118       sha1_process_round(a, b, c, d, e, cur_k, cur_w, t2, round);
6119     }
6120 
6121     // compute the intermediate hash value
6122     sha1_calculate_im_hash(a, b, c, d, e, prev_ab, prev_cd, prev_e);
6123 
6124     if (multi_block) {
6125       int64_t block_bytes = 16 * 4;
6126       __ addi(buf, buf, block_bytes);
6127 
6128       __ bge(limit, buf, L_sha1_loop, true);
6129     }
6130 
6131     // store back the state.
6132     __ zext(a, a, 32);
6133     __ slli(b, b, 32);
6134     __ orr(a, a, b);
6135     __ sd(a, Address(state, 0));
6136     __ zext(c, c, 32);
6137     __ slli(d, d, 32);
6138     __ orr(c, c, d);
6139     __ sd(c, Address(state, 8));
6140     __ sw(e, Address(state, 16));
6141 
6142     // return offset
6143     if (multi_block) {
6144       __ sub(c_rarg0, buf, src);
6145     }
6146 
6147     __ pop_reg(saved_regs, sp);
6148 
6149     __ leave();
6150     __ ret();
6151 
6152     return (address) start;
6153   }
6154 
6155   /**
6156    * vector registers:
6157    *   input VectorRegister's:  intputV1-V3, for m2 they could be v2, v4, v6, for m1 they could be v1, v2, v3
6158    *   index VectorRegister's:  idxV1-V4, for m2 they could be v8, v10, v12, v14, for m1 they could be v4, v5, v6, v7
6159    *   output VectorRegister's: outputV1-V4, for m2 they could be v16, v18, v20, v22, for m1 they could be v8, v9, v10, v11
6160    *
6161    * NOTE: each field will occupy a vector register group
6162    */
6163   void base64_vector_encode_round(Register src, Register dst, Register codec,
6164                     Register size, Register stepSrc, Register stepDst,
6165                     VectorRegister inputV1, VectorRegister inputV2, VectorRegister inputV3,
6166                     VectorRegister idxV1, VectorRegister idxV2, VectorRegister idxV3, VectorRegister idxV4,
6167                     VectorRegister outputV1, VectorRegister outputV2, VectorRegister outputV3, VectorRegister outputV4,
6168                     Assembler::LMUL lmul) {
6169     // set vector register type/len
6170     __ vsetvli(x0, size, Assembler::e8, lmul);
6171 
6172     // segmented load src into v registers: mem(src) => vr(3)
6173     __ vlseg3e8_v(inputV1, src);
6174 
6175     // src = src + register_group_len_bytes * 3
6176     __ add(src, src, stepSrc);
6177 
6178     // encoding
6179     //   1. compute index into lookup table: vr(3) => vr(4)
6180     __ vsrl_vi(idxV1, inputV1, 2);
6181 
6182     __ vsrl_vi(idxV2, inputV2, 2);
6183     __ vsll_vi(inputV1, inputV1, 6);
6184     __ vor_vv(idxV2, idxV2, inputV1);
6185     __ vsrl_vi(idxV2, idxV2, 2);
6186 
6187     __ vsrl_vi(idxV3, inputV3, 4);
6188     __ vsll_vi(inputV2, inputV2, 4);
6189     __ vor_vv(idxV3, inputV2, idxV3);
6190     __ vsrl_vi(idxV3, idxV3, 2);
6191 
6192     __ vsll_vi(idxV4, inputV3, 2);
6193     __ vsrl_vi(idxV4, idxV4, 2);
6194 
6195     //   2. indexed load: vr(4) => vr(4)
6196     __ vluxei8_v(outputV1, codec, idxV1);
6197     __ vluxei8_v(outputV2, codec, idxV2);
6198     __ vluxei8_v(outputV3, codec, idxV3);
6199     __ vluxei8_v(outputV4, codec, idxV4);
6200 
6201     // segmented store encoded data in v registers back to dst: vr(4) => mem(dst)
6202     __ vsseg4e8_v(outputV1, dst);
6203 
6204     // dst = dst + register_group_len_bytes * 4
6205     __ add(dst, dst, stepDst);
6206   }
6207 
6208   /**
6209    *  void j.u.Base64.Encoder.encodeBlock(byte[] src, int sp, int sl, byte[] dst, int dp, boolean isURL)
6210    *
6211    *  Input arguments:
6212    *  c_rarg0   - src, source array
6213    *  c_rarg1   - sp, src start offset
6214    *  c_rarg2   - sl, src end offset
6215    *  c_rarg3   - dst, dest array
6216    *  c_rarg4   - dp, dst start offset
6217    *  c_rarg5   - isURL, Base64 or URL character set
6218    */
6219   address generate_base64_encodeBlock() {
6220     alignas(64) static const char toBase64[64] = {
6221       'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
6222       'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z',
6223       'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm',
6224       'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z',
6225       '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '+', '/'
6226     };
6227 
6228     alignas(64) static const char toBase64URL[64] = {
6229       'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
6230       'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z',
6231       'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm',
6232       'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z',
6233       '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '-', '_'
6234     };
6235 
6236     __ align(CodeEntryAlignment);
6237     StubId stub_id = StubId::stubgen_base64_encodeBlock_id;
6238     StubCodeMark mark(this, stub_id);
6239     address start = __ pc();
6240     __ enter();
6241 
6242     Register src    = c_rarg0;
6243     Register soff   = c_rarg1;
6244     Register send   = c_rarg2;
6245     Register dst    = c_rarg3;
6246     Register doff   = c_rarg4;
6247     Register isURL  = c_rarg5;
6248 
6249     Register codec  = c_rarg6;
6250     Register length = c_rarg7; // total length of src data in bytes
6251 
6252     Label ProcessData, Exit;
6253 
6254     // length should be multiple of 3
6255     __ sub(length, send, soff);
6256     // real src/dst to process data
6257     __ add(src, src, soff);
6258     __ add(dst, dst, doff);
6259 
6260     // load the codec base address
6261     __ la(codec, ExternalAddress((address) toBase64));
6262     __ beqz(isURL, ProcessData);
6263     __ la(codec, ExternalAddress((address) toBase64URL));
6264     __ BIND(ProcessData);
6265 
6266     // vector version
6267     if (UseRVV) {
6268       Label ProcessM2, ProcessM1, ProcessScalar;
6269 
6270       Register size      = soff;
6271       Register stepSrcM1 = send;
6272       Register stepSrcM2 = doff;
6273       Register stepDst   = isURL;
6274 
6275       __ mv(size, MaxVectorSize * 2);
6276       __ mv(stepSrcM1, MaxVectorSize * 3);
6277       __ slli(stepSrcM2, stepSrcM1, 1);
6278       __ mv(stepDst, MaxVectorSize * 2 * 4);
6279 
6280       __ blt(length, stepSrcM2, ProcessM1);
6281 
6282       __ BIND(ProcessM2);
6283       base64_vector_encode_round(src, dst, codec,
6284                     size, stepSrcM2, stepDst,
6285                     v2, v4, v6,         // inputs
6286                     v8, v10, v12, v14,  // indexes
6287                     v16, v18, v20, v22, // outputs
6288                     Assembler::m2);
6289 
6290       __ sub(length, length, stepSrcM2);
6291       __ bge(length, stepSrcM2, ProcessM2);
6292 
6293       __ BIND(ProcessM1);
6294       __ blt(length, stepSrcM1, ProcessScalar);
6295 
6296       __ srli(size, size, 1);
6297       __ srli(stepDst, stepDst, 1);
6298       base64_vector_encode_round(src, dst, codec,
6299                     size, stepSrcM1, stepDst,
6300                     v1, v2, v3,         // inputs
6301                     v4, v5, v6, v7,     // indexes
6302                     v8, v9, v10, v11,   // outputs
6303                     Assembler::m1);
6304       __ sub(length, length, stepSrcM1);
6305 
6306       __ BIND(ProcessScalar);
6307     }
6308 
6309     // scalar version
6310     {
6311       Register byte1 = soff, byte0 = send, byte2 = doff;
6312       Register combined24Bits = isURL;
6313 
6314       __ beqz(length, Exit);
6315 
6316       Label ScalarLoop;
6317       __ BIND(ScalarLoop);
6318       {
6319         // plain:   [byte0[7:0] : byte1[7:0] : byte2[7:0]] =>
6320         // encoded: [byte0[7:2] : byte0[1:0]+byte1[7:4] : byte1[3:0]+byte2[7:6] : byte2[5:0]]
6321 
6322         // load 3 bytes src data
6323         __ lbu(byte0, Address(src, 0));
6324         __ lbu(byte1, Address(src, 1));
6325         __ lbu(byte2, Address(src, 2));
6326         __ addi(src, src, 3);
6327 
6328         // construct 24 bits from 3 bytes
6329         __ slliw(byte0, byte0, 16);
6330         __ slliw(byte1, byte1, 8);
6331         __ orr(combined24Bits, byte0, byte1);
6332         __ orr(combined24Bits, combined24Bits, byte2);
6333 
6334         // get codec index and encode(ie. load from codec by index)
6335         __ slliw(byte0, combined24Bits, 8);
6336         __ srliw(byte0, byte0, 26);
6337         __ add(byte0, codec, byte0);
6338         __ lbu(byte0, byte0);
6339 
6340         __ slliw(byte1, combined24Bits, 14);
6341         __ srliw(byte1, byte1, 26);
6342         __ add(byte1, codec, byte1);
6343         __ lbu(byte1, byte1);
6344 
6345         __ slliw(byte2, combined24Bits, 20);
6346         __ srliw(byte2, byte2, 26);
6347         __ add(byte2, codec, byte2);
6348         __ lbu(byte2, byte2);
6349 
6350         __ andi(combined24Bits, combined24Bits, 0x3f);
6351         __ add(combined24Bits, codec, combined24Bits);
6352         __ lbu(combined24Bits, combined24Bits);
6353 
6354         // store 4 bytes encoded data
6355         __ sb(byte0, Address(dst, 0));
6356         __ sb(byte1, Address(dst, 1));
6357         __ sb(byte2, Address(dst, 2));
6358         __ sb(combined24Bits, Address(dst, 3));
6359 
6360         __ subi(length, length, 3);
6361         __ addi(dst, dst, 4);
6362         // loop back
6363         __ bnez(length, ScalarLoop);
6364       }
6365     }
6366 
6367     __ BIND(Exit);
6368 
6369     __ leave();
6370     __ ret();
6371 
6372     return (address) start;
6373   }
6374 
6375   /**
6376    * vector registers:
6377    * input VectorRegister's:  intputV1-V4, for m2 they could be v2, v4, v6, for m1 they could be v2, v4, v6, v8
6378    * index VectorRegister's:  idxV1-V3, for m2 they could be v8, v10, v12, v14, for m1 they could be v10, v12, v14, v16
6379    * output VectorRegister's: outputV1-V4, for m2 they could be v16, v18, v20, v22, for m1 they could be v18, v20, v22
6380    *
6381    * NOTE: each field will occupy a single vector register group
6382    */
6383   void base64_vector_decode_round(Register src, Register dst, Register codec,
6384                     Register size, Register stepSrc, Register stepDst, Register failedIdx,
6385                     VectorRegister inputV1, VectorRegister inputV2, VectorRegister inputV3, VectorRegister inputV4,
6386                     VectorRegister idxV1, VectorRegister idxV2, VectorRegister idxV3, VectorRegister idxV4,
6387                     VectorRegister outputV1, VectorRegister outputV2, VectorRegister outputV3,
6388                     Assembler::LMUL lmul) {
6389     // set vector register type/len
6390     __ vsetvli(x0, size, Assembler::e8, lmul, Assembler::ma, Assembler::ta);
6391 
6392     // segmented load src into v registers: mem(src) => vr(4)
6393     __ vlseg4e8_v(inputV1, src);
6394 
6395     // src = src + register_group_len_bytes * 4
6396     __ add(src, src, stepSrc);
6397 
6398     // decoding
6399     //   1. indexed load: vr(4) => vr(4)
6400     __ vluxei8_v(idxV1, codec, inputV1);
6401     __ vluxei8_v(idxV2, codec, inputV2);
6402     __ vluxei8_v(idxV3, codec, inputV3);
6403     __ vluxei8_v(idxV4, codec, inputV4);
6404 
6405     //   2. check wrong data
6406     __ vor_vv(outputV1, idxV1, idxV2);
6407     __ vor_vv(outputV2, idxV3, idxV4);
6408     __ vor_vv(outputV1, outputV1, outputV2);
6409     __ vmseq_vi(v0, outputV1, -1);
6410     __ vfirst_m(failedIdx, v0);
6411     Label NoFailure, FailureAtIdx0;
6412     // valid value can only be -1 when < 0
6413     __ bltz(failedIdx, NoFailure);
6414     // when the first data (at index 0) fails, no need to process data anymore
6415     __ beqz(failedIdx, FailureAtIdx0);
6416     __ vsetvli(x0, failedIdx, Assembler::e8, lmul, Assembler::mu, Assembler::tu);
6417     __ slli(stepDst, failedIdx, 1);
6418     __ add(stepDst, failedIdx, stepDst);
6419     __ BIND(NoFailure);
6420 
6421     //   3. compute the decoded data: vr(4) => vr(3)
6422     __ vsll_vi(idxV1, idxV1, 2);
6423     __ vsrl_vi(outputV1, idxV2, 4);
6424     __ vor_vv(outputV1, outputV1, idxV1);
6425 
6426     __ vsll_vi(idxV2, idxV2, 4);
6427     __ vsrl_vi(outputV2, idxV3, 2);
6428     __ vor_vv(outputV2, outputV2, idxV2);
6429 
6430     __ vsll_vi(idxV3, idxV3, 6);
6431     __ vor_vv(outputV3, idxV4, idxV3);
6432 
6433     // segmented store encoded data in v registers back to dst: vr(3) => mem(dst)
6434     __ vsseg3e8_v(outputV1, dst);
6435 
6436     // dst = dst + register_group_len_bytes * 3
6437     __ add(dst, dst, stepDst);
6438     __ BIND(FailureAtIdx0);
6439   }
6440 
6441   /**
6442    * int j.u.Base64.Decoder.decodeBlock(byte[] src, int sp, int sl, byte[] dst, int dp, boolean isURL, boolean isMIME)
6443    *
6444    *  Input arguments:
6445    *  c_rarg0   - src, source array
6446    *  c_rarg1   - sp, src start offset
6447    *  c_rarg2   - sl, src end offset
6448    *  c_rarg3   - dst, dest array
6449    *  c_rarg4   - dp, dst start offset
6450    *  c_rarg5   - isURL, Base64 or URL character set
6451    *  c_rarg6   - isMIME, Decoding MIME block
6452    */
6453   address generate_base64_decodeBlock() {
6454 
6455     static const uint8_t fromBase64[256] = {
6456         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6457         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6458         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,  62u, 255u, 255u, 255u,  63u,
6459         52u,  53u,  54u,  55u,  56u,  57u,  58u,  59u,  60u,  61u, 255u, 255u, 255u, 255u, 255u, 255u,
6460         255u,   0u,   1u,   2u,   3u,   4u,   5u,   6u,   7u,   8u,   9u,  10u,  11u,  12u,  13u,  14u,
6461         15u,  16u,  17u,  18u,  19u,  20u,  21u,  22u,  23u,  24u,  25u, 255u, 255u, 255u, 255u, 255u,
6462         255u,  26u,  27u,  28u,  29u,  30u,  31u,  32u,  33u,  34u,  35u,  36u,  37u,  38u,  39u,  40u,
6463         41u,  42u,  43u,  44u,  45u,  46u,  47u,  48u,  49u,  50u,  51u, 255u, 255u, 255u, 255u, 255u,
6464         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6465         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6466         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6467         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6468         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6469         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6470         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6471         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6472     };
6473 
6474     static const uint8_t fromBase64URL[256] = {
6475         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6476         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6477         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,  62u, 255u, 255u,
6478         52u,  53u,  54u,  55u,  56u,  57u,  58u,  59u,  60u,  61u, 255u, 255u, 255u, 255u, 255u, 255u,
6479         255u,   0u,   1u,   2u,   3u,   4u,   5u,   6u,   7u,   8u,   9u,  10u,  11u,  12u,  13u,  14u,
6480         15u,  16u,  17u,  18u,  19u,  20u,  21u,  22u,  23u,  24u,  25u, 255u, 255u, 255u, 255u,  63u,
6481         255u,  26u,  27u,  28u,  29u,  30u,  31u,  32u,  33u,  34u,  35u,  36u,  37u,  38u,  39u,  40u,
6482         41u,  42u,  43u,  44u,  45u,  46u,  47u,  48u,  49u,  50u,  51u, 255u, 255u, 255u, 255u, 255u,
6483         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6484         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6485         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6486         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6487         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6488         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6489         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6490         255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6491     };
6492 
6493     __ align(CodeEntryAlignment);
6494     StubId stub_id = StubId::stubgen_base64_decodeBlock_id;
6495     StubCodeMark mark(this, stub_id);
6496     address start = __ pc();
6497     __ enter();
6498 
6499     Register src    = c_rarg0;
6500     Register soff   = c_rarg1;
6501     Register send   = c_rarg2;
6502     Register dst    = c_rarg3;
6503     Register doff   = c_rarg4;
6504     Register isURL  = c_rarg5;
6505     Register isMIME = c_rarg6;
6506 
6507     Register codec     = c_rarg7;
6508     Register dstBackup = t6;
6509     Register length    = t3;     // total length of src data in bytes
6510 
6511     Label ProcessData, Exit;
6512     Label ProcessScalar, ScalarLoop;
6513 
6514     // passed in length (send - soff) is guaranteed to be > 4,
6515     // and in this intrinsic we only process data of length in multiple of 4,
6516     // it's not guaranteed to be multiple of 4 by java level, so do it explicitly
6517     __ sub(length, send, soff);
6518     __ andi(length, length, -4);
6519     // real src/dst to process data
6520     __ add(src, src, soff);
6521     __ add(dst, dst, doff);
6522     // backup of dst, used to calculate the return value at exit
6523     __ mv(dstBackup, dst);
6524 
6525     // load the codec base address
6526     __ la(codec, ExternalAddress((address) fromBase64));
6527     __ beqz(isURL, ProcessData);
6528     __ la(codec, ExternalAddress((address) fromBase64URL));
6529     __ BIND(ProcessData);
6530 
6531     // vector version
6532     if (UseRVV) {
6533       // for MIME case, it has a default length limit of 76 which could be
6534       // different(smaller) from (send - soff), so in MIME case, we go through
6535       // the scalar code path directly.
6536       __ bnez(isMIME, ScalarLoop);
6537 
6538       Label ProcessM1, ProcessM2;
6539 
6540       Register failedIdx = soff;
6541       Register stepSrcM1 = send;
6542       Register stepSrcM2 = doff;
6543       Register stepDst   = isURL;
6544       Register size      = t4;
6545 
6546       __ mv(size, MaxVectorSize * 2);
6547       __ mv(stepSrcM1, MaxVectorSize * 4);
6548       __ slli(stepSrcM2, stepSrcM1, 1);
6549       __ mv(stepDst, MaxVectorSize * 2 * 3);
6550 
6551       __ blt(length, stepSrcM2, ProcessM1);
6552 
6553 
6554       // Assembler::m2
6555       __ BIND(ProcessM2);
6556       base64_vector_decode_round(src, dst, codec,
6557                     size, stepSrcM2, stepDst, failedIdx,
6558                     v2, v4, v6, v8,      // inputs
6559                     v10, v12, v14, v16,  // indexes
6560                     v18, v20, v22,       // outputs
6561                     Assembler::m2);
6562       __ sub(length, length, stepSrcM2);
6563 
6564       // error check
6565       // valid value of failedIdx can only be -1 when < 0
6566       __ bgez(failedIdx, Exit);
6567 
6568       __ bge(length, stepSrcM2, ProcessM2);
6569 
6570 
6571       // Assembler::m1
6572       __ BIND(ProcessM1);
6573       __ blt(length, stepSrcM1, ProcessScalar);
6574 
6575       __ srli(size, size, 1);
6576       __ srli(stepDst, stepDst, 1);
6577       base64_vector_decode_round(src, dst, codec,
6578                     size, stepSrcM1, stepDst, failedIdx,
6579                     v1, v2, v3, v4,      // inputs
6580                     v5, v6, v7, v8,      // indexes
6581                     v9, v10, v11,        // outputs
6582                     Assembler::m1);
6583       __ sub(length, length, stepSrcM1);
6584 
6585       // error check
6586       // valid value of failedIdx can only be -1 when < 0
6587       __ bgez(failedIdx, Exit);
6588 
6589       __ BIND(ProcessScalar);
6590       __ beqz(length, Exit);
6591     }
6592 
6593     // scalar version
6594     {
6595       Register byte0 = soff, byte1 = send, byte2 = doff, byte3 = isURL;
6596       Register combined32Bits = t4;
6597 
6598       // encoded:   [byte0[5:0] : byte1[5:0] : byte2[5:0]] : byte3[5:0]] =>
6599       // plain:     [byte0[5:0]+byte1[5:4] : byte1[3:0]+byte2[5:2] : byte2[1:0]+byte3[5:0]]
6600       __ BIND(ScalarLoop);
6601 
6602       // load 4 bytes encoded src data
6603       __ lbu(byte0, Address(src, 0));
6604       __ lbu(byte1, Address(src, 1));
6605       __ lbu(byte2, Address(src, 2));
6606       __ lbu(byte3, Address(src, 3));
6607       __ addi(src, src, 4);
6608 
6609       // get codec index and decode (ie. load from codec by index)
6610       __ add(byte0, codec, byte0);
6611       __ add(byte1, codec, byte1);
6612       __ lb(byte0, Address(byte0, 0));
6613       __ lb(byte1, Address(byte1, 0));
6614       __ add(byte2, codec, byte2);
6615       __ add(byte3, codec, byte3);
6616       __ lb(byte2, Address(byte2, 0));
6617       __ lb(byte3, Address(byte3, 0));
6618       __ slliw(byte0, byte0, 18);
6619       __ slliw(byte1, byte1, 12);
6620       __ orr(byte0, byte0, byte1);
6621       __ orr(byte0, byte0, byte3);
6622       __ slliw(byte2, byte2, 6);
6623       // For performance consideration, `combined32Bits` is constructed for 2 purposes at the same time,
6624       //  1. error check below
6625       //  2. decode below
6626       __ orr(combined32Bits, byte0, byte2);
6627 
6628       // error check
6629       __ bltz(combined32Bits, Exit);
6630 
6631       // store 3 bytes decoded data
6632       __ sraiw(byte0, combined32Bits, 16);
6633       __ sraiw(byte1, combined32Bits, 8);
6634       __ sb(byte0, Address(dst, 0));
6635       __ sb(byte1, Address(dst, 1));
6636       __ sb(combined32Bits, Address(dst, 2));
6637 
6638       __ subi(length, length, 4);
6639       __ addi(dst, dst, 3);
6640       // loop back
6641       __ bnez(length, ScalarLoop);
6642     }
6643 
6644     __ BIND(Exit);
6645     __ sub(c_rarg0, dst, dstBackup);
6646 
6647     __ leave();
6648     __ ret();
6649 
6650     return (address) start;
6651   }
6652 
6653   void adler32_process_bytes(Register buff, Register s1, Register s2, VectorRegister vtable,
6654     VectorRegister vzero, VectorRegister vbytes, VectorRegister vs1acc, VectorRegister vs2acc,
6655     Register temp0, Register temp1, Register temp2,  Register temp3,
6656     VectorRegister vtemp1, VectorRegister vtemp2, int step, Assembler::LMUL lmul) {
6657 
6658     assert((lmul == Assembler::m4 && step == 64) ||
6659            (lmul == Assembler::m2 && step == 32) ||
6660            (lmul == Assembler::m1 && step == 16),
6661            "LMUL should be aligned with step: m4 and 64, m2 and 32 or m1 and 16");
6662     // Below is function for calculating Adler32 checksum with 64-, 32- or 16-byte step. LMUL=m4, m2 or m1 is used.
6663     // The results are in v12, v13, ..., v22, v23. Example below is for 64-byte step case.
6664     // We use b1, b2, ..., b64 to denote the 64 bytes loaded in each iteration.
6665     // In non-vectorized code, we update s1 and s2 as:
6666     //   s1 <- s1 + b1
6667     //   s2 <- s2 + s1
6668     //   s1 <- s1 + b2
6669     //   s2 <- s2 + b1
6670     //   ...
6671     //   s1 <- s1 + b64
6672     //   s2 <- s2 + s1
6673     // Putting above assignments together, we have:
6674     //   s1_new = s1 + b1 + b2 + ... + b64
6675     //   s2_new = s2 + (s1 + b1) + (s1 + b1 + b2) + ... + (s1 + b1 + b2 + ... + b64) =
6676     //          = s2 + s1 * 64 + (b1 * 64 + b2 * 63 + ... + b64 * 1) =
6677     //          = s2 + s1 * 64 + (b1, b2, ... b64) dot (64, 63, ... 1)
6678 
6679     __ mv(temp3, step);
6680     // Load data
6681     __ vsetvli(temp0, temp3, Assembler::e8, lmul);
6682     __ vle8_v(vbytes, buff);
6683     __ addi(buff, buff, step);
6684 
6685     // Upper bound reduction sum for s1_new:
6686     // 0xFF * 64 = 0x3FC0, so:
6687     // 1. Need to do vector-widening reduction sum
6688     // 2. It is safe to perform sign-extension during vmv.x.s with 16-bits elements
6689     __ vwredsumu_vs(vs1acc, vbytes, vzero);
6690     // Multiplication for s2_new
6691     __ vwmulu_vv(vs2acc, vtable, vbytes);
6692 
6693     // s2 = s2 + s1 * log2(step)
6694     __ slli(temp1, s1, exact_log2(step));
6695     __ add(s2, s2, temp1);
6696 
6697     // Summing up calculated results for s2_new
6698     if (MaxVectorSize > 16) {
6699       __ vsetvli(temp0, temp3, Assembler::e16, lmul);
6700     } else {
6701       // Half of vector-widening multiplication result is in successor of vs2acc
6702       // group for vlen == 16, in which case we need to double vector register
6703       // group width in order to reduction sum all of them
6704       Assembler::LMUL lmulx2 = (lmul == Assembler::m1) ? Assembler::m2 :
6705                                (lmul == Assembler::m2) ? Assembler::m4 : Assembler::m8;
6706       __ vsetvli(temp0, temp3, Assembler::e16, lmulx2);
6707     }
6708     // Upper bound for reduction sum:
6709     // 0xFF * (64 + 63 + ... + 2 + 1) = 0x817E0 max for whole register group, so:
6710     // 1. Need to do vector-widening reduction sum
6711     // 2. It is safe to perform sign-extension during vmv.x.s with 32-bits elements
6712     __ vwredsumu_vs(vtemp1, vs2acc, vzero);
6713 
6714     // Extracting results for:
6715     // s1_new
6716     __ vmv_x_s(temp0, vs1acc);
6717     __ add(s1, s1, temp0);
6718     // s2_new
6719     __ vsetvli(temp0, temp3, Assembler::e32, Assembler::m1);
6720     __ vmv_x_s(temp1, vtemp1);
6721     __ add(s2, s2, temp1);
6722   }
6723 
6724   /***
6725    *  int java.util.zip.Adler32.updateBytes(int adler, byte[] b, int off, int len)
6726    *
6727    *  Arguments:
6728    *
6729    *  Inputs:
6730    *   c_rarg0   - int   adler
6731    *   c_rarg1   - byte* buff (b + off)
6732    *   c_rarg2   - int   len
6733    *
6734    *  Output:
6735    *   c_rarg0   - int adler result
6736    */
6737   address generate_updateBytesAdler32() {
6738     __ align(CodeEntryAlignment);
6739     StubId stub_id = StubId::stubgen_updateBytesAdler32_id;
6740     StubCodeMark mark(this, stub_id);
6741     address start = __ pc();
6742 
6743     Label L_nmax, L_nmax_loop, L_nmax_loop_entry, L_by16, L_by16_loop,
6744       L_by16_loop_unroll, L_by1_loop, L_do_mod, L_combine, L_by1;
6745 
6746     // Aliases
6747     Register adler  = c_rarg0;
6748     Register s1     = c_rarg0;
6749     Register s2     = c_rarg3;
6750     Register buff   = c_rarg1;
6751     Register len    = c_rarg2;
6752     Register nmax  = c_rarg4;
6753     Register base  = c_rarg5;
6754     Register count = c_rarg6;
6755     Register temp0 = t3;
6756     Register temp1 = t4;
6757     Register temp2 = t5;
6758     Register temp3 = t6;
6759 
6760     VectorRegister vzero = v31;
6761     VectorRegister vbytes = v8; // group: v8, v9, v10, v11
6762     VectorRegister vs1acc = v12; // group: v12, v13, v14, v15
6763     VectorRegister vs2acc = v16; // group: v16, v17, v18, v19, v20, v21, v22, v23
6764     VectorRegister vtable_64 = v24; // group: v24, v25, v26, v27
6765     VectorRegister vtable_32 = v4; // group: v4, v5
6766     VectorRegister vtable_16 = v30;
6767     VectorRegister vtemp1 = v28;
6768     VectorRegister vtemp2 = v29;
6769 
6770     // Max number of bytes we can process before having to take the mod
6771     // 0x15B0 is 5552 in decimal, the largest n such that 255n(n+1)/2 + (n+1)(BASE-1) <= 2^32-1
6772     const uint64_t BASE = 0xfff1;
6773     const uint64_t NMAX = 0x15B0;
6774 
6775     // Loops steps
6776     int step_64 = 64;
6777     int step_32 = 32;
6778     int step_16 = 16;
6779     int step_1  = 1;
6780 
6781     __ enter(); // Required for proper stackwalking of RuntimeStub frame
6782     __ mv(temp1, 64);
6783     __ vsetvli(temp0, temp1, Assembler::e8, Assembler::m4);
6784 
6785     // Generating accumulation coefficients for further calculations
6786     // vtable_64:
6787     __ vid_v(vtemp1);
6788     __ vrsub_vx(vtable_64, vtemp1, temp1);
6789     // vtable_64 group now contains { 0x40, 0x3f, 0x3e, ..., 0x3, 0x2, 0x1 }
6790 
6791     // vtable_32:
6792     __ mv(temp1, 32);
6793     __ vsetvli(temp0, temp1, Assembler::e8, Assembler::m2);
6794     __ vid_v(vtemp1);
6795     __ vrsub_vx(vtable_32, vtemp1, temp1);
6796     // vtable_32 group now contains { 0x20, 0x1f, 0x1e, ..., 0x3, 0x2, 0x1 }
6797 
6798     __ vsetivli(temp0, 16, Assembler::e8, Assembler::m1);
6799     // vtable_16:
6800     __ mv(temp1, 16);
6801     __ vid_v(vtemp1);
6802     __ vrsub_vx(vtable_16, vtemp1, temp1);
6803     // vtable_16 now contains { 0x10, 0xf, 0xe, ..., 0x3, 0x2, 0x1 }
6804 
6805     __ vmv_v_i(vzero, 0);
6806 
6807     __ mv(base, BASE);
6808     __ mv(nmax, NMAX);
6809 
6810     // s1 is initialized to the lower 16 bits of adler
6811     // s2 is initialized to the upper 16 bits of adler
6812     __ srliw(s2, adler, 16); // s2 = ((adler >> 16) & 0xffff)
6813     __ zext(s1, adler, 16); // s1 = (adler & 0xffff)
6814 
6815     // The pipelined loop needs at least 16 elements for 1 iteration
6816     // It does check this, but it is more effective to skip to the cleanup loop
6817     __ mv(temp0, step_16);
6818     __ bgeu(len, temp0, L_nmax);
6819     __ beqz(len, L_combine);
6820 
6821     // Jumping to L_by1_loop
6822     __ subi(len, len, step_1);
6823     __ j(L_by1_loop);
6824 
6825   __ bind(L_nmax);
6826     __ sub(len, len, nmax);
6827     __ subi(count, nmax, 16);
6828     __ bltz(len, L_by16);
6829 
6830   // Align L_nmax loop by 64
6831   __ bind(L_nmax_loop_entry);
6832     __ subi(count, count, 32);
6833 
6834   __ bind(L_nmax_loop);
6835     adler32_process_bytes(buff, s1, s2, vtable_64, vzero,
6836       vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6837       vtemp1, vtemp2, step_64, Assembler::m4);
6838     __ subi(count, count, step_64);
6839     __ bgtz(count, L_nmax_loop);
6840 
6841     // There are three iterations left to do
6842     adler32_process_bytes(buff, s1, s2, vtable_32, vzero,
6843       vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6844       vtemp1, vtemp2, step_32, Assembler::m2);
6845     adler32_process_bytes(buff, s1, s2, vtable_16, vzero,
6846       vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6847       vtemp1, vtemp2, step_16, Assembler::m1);
6848 
6849     // s1 = s1 % BASE
6850     __ remuw(s1, s1, base);
6851     // s2 = s2 % BASE
6852     __ remuw(s2, s2, base);
6853 
6854     __ sub(len, len, nmax);
6855     __ subi(count, nmax, 16);
6856     __ bgez(len, L_nmax_loop_entry);
6857 
6858   __ bind(L_by16);
6859     __ add(len, len, count);
6860     __ bltz(len, L_by1);
6861     // Trying to unroll
6862     __ mv(temp3, step_64);
6863     __ blt(len, temp3, L_by16_loop);
6864 
6865   __ bind(L_by16_loop_unroll);
6866     adler32_process_bytes(buff, s1, s2, vtable_64, vzero,
6867       vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6868       vtemp1, vtemp2, step_64, Assembler::m4);
6869     __ subi(len, len, step_64);
6870     // By now the temp3 should still be 64
6871     __ bge(len, temp3, L_by16_loop_unroll);
6872 
6873   __ bind(L_by16_loop);
6874     adler32_process_bytes(buff, s1, s2, vtable_16, vzero,
6875       vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6876       vtemp1, vtemp2, step_16, Assembler::m1);
6877     __ subi(len, len, step_16);
6878     __ bgez(len, L_by16_loop);
6879 
6880   __ bind(L_by1);
6881     __ addi(len, len, 15);
6882     __ bltz(len, L_do_mod);
6883 
6884   __ bind(L_by1_loop);
6885     __ lbu(temp0, Address(buff, 0));
6886     __ addi(buff, buff, step_1);
6887     __ add(s1, temp0, s1);
6888     __ add(s2, s2, s1);
6889     __ subi(len, len, step_1);
6890     __ bgez(len, L_by1_loop);
6891 
6892   __ bind(L_do_mod);
6893     // s1 = s1 % BASE
6894     __ remuw(s1, s1, base);
6895     // s2 = s2 % BASE
6896     __ remuw(s2, s2, base);
6897 
6898     // Combine lower bits and higher bits
6899     // adler = s1 | (s2 << 16)
6900   __ bind(L_combine);
6901     __ slli(s2, s2, 16);
6902     __ orr(s1, s1, s2);
6903 
6904     __ leave(); // Required for proper stackwalking of RuntimeStub frame
6905     __ ret();
6906 
6907     return start;
6908   }
6909 
6910 #endif // COMPILER2
6911 
6912   // x10 = input (float16)
6913   // f10 = result (float)
6914   // t1  = temporary register
6915   address generate_float16ToFloat() {
6916     __ align(CodeEntryAlignment);
6917     StubId stub_id = StubId::stubgen_hf2f_id;
6918     StubCodeMark mark(this, stub_id);
6919     address entry = __ pc();
6920     BLOCK_COMMENT("float16ToFloat:");
6921 
6922     FloatRegister dst = f10;
6923     Register src = x10;
6924     Label NaN_SLOW;
6925 
6926     assert(VM_Version::supports_float16_float_conversion(), "must");
6927 
6928     // On riscv, NaN needs a special process as fcvt does not work in that case.
6929     // On riscv, Inf does not need a special process as fcvt can handle it correctly.
6930     // but we consider to get the slow path to process NaN and Inf at the same time,
6931     // as both of them are rare cases, and if we try to get the slow path to handle
6932     // only NaN case it would sacrifise the performance for normal cases,
6933     // i.e. non-NaN and non-Inf cases.
6934 
6935     // check whether it's a NaN or +/- Inf.
6936     __ mv(t0, 0x7c00);
6937     __ andr(t1, src, t0);
6938     // jump to stub processing NaN and Inf cases.
6939     __ beq(t0, t1, NaN_SLOW);
6940 
6941     // non-NaN or non-Inf cases, just use built-in instructions.
6942     __ fmv_h_x(dst, src);
6943     __ fcvt_s_h(dst, dst);
6944     __ ret();
6945 
6946     __ bind(NaN_SLOW);
6947     // following instructions mainly focus on NaN, as riscv does not handle
6948     // NaN well with fcvt, but the code also works for Inf at the same time.
6949 
6950     // construct a NaN in 32 bits from the NaN in 16 bits,
6951     // we need the payloads of non-canonical NaNs to be preserved.
6952     __ mv(t1, 0x7f800000);
6953     // sign-bit was already set via sign-extension if necessary.
6954     __ slli(t0, src, 13);
6955     __ orr(t1, t0, t1);
6956     __ fmv_w_x(dst, t1);
6957 
6958     __ ret();
6959     return entry;
6960   }
6961 
6962   // f10 = input (float)
6963   // x10 = result (float16)
6964   // f11 = temporary float register
6965   // t1  = temporary register
6966   address generate_floatToFloat16() {
6967     __ align(CodeEntryAlignment);
6968     StubId stub_id = StubId::stubgen_f2hf_id;
6969     StubCodeMark mark(this, stub_id);
6970     address entry = __ pc();
6971     BLOCK_COMMENT("floatToFloat16:");
6972 
6973     Register dst = x10;
6974     FloatRegister src = f10, ftmp = f11;
6975     Label NaN_SLOW;
6976 
6977     assert(VM_Version::supports_float16_float_conversion(), "must");
6978 
6979     // On riscv, NaN needs a special process as fcvt does not work in that case.
6980 
6981     // check whether it's a NaN.
6982     // replace fclass with feq as performance optimization.
6983     __ feq_s(t0, src, src);
6984     // jump to stub processing NaN cases.
6985     __ beqz(t0, NaN_SLOW);
6986 
6987     // non-NaN cases, just use built-in instructions.
6988     __ fcvt_h_s(ftmp, src);
6989     __ fmv_x_h(dst, ftmp);
6990     __ ret();
6991 
6992     __ bind(NaN_SLOW);
6993 
6994     __ float_to_float16_NaN(dst, src, t0, t1);
6995 
6996     __ ret();
6997     return entry;
6998   }
6999 
7000 #ifdef COMPILER2
7001 
7002 static const int64_t right_2_bits = right_n_bits(2);
7003 static const int64_t right_3_bits = right_n_bits(3);
7004 
7005   // In sun.security.util.math.intpoly.IntegerPolynomial1305, integers
7006   // are represented as long[5], with BITS_PER_LIMB = 26.
7007   // Pack five 26-bit limbs into three 64-bit registers.
7008   void poly1305_pack_26(Register dest0, Register dest1, Register dest2, Register src, Register tmp1, Register tmp2) {
7009     assert_different_registers(dest0, dest1, dest2, src, tmp1, tmp2);
7010 
7011     // The goal is to have 128-bit value in dest2:dest1:dest0
7012     __ ld(dest0, Address(src, 0));    // 26 bits in dest0
7013 
7014     __ ld(tmp1, Address(src, sizeof(jlong)));
7015     __ slli(tmp1, tmp1, 26);
7016     __ add(dest0, dest0, tmp1);       // 52 bits in dest0
7017 
7018     __ ld(tmp2, Address(src, 2 * sizeof(jlong)));
7019     __ slli(tmp1, tmp2, 52);
7020     __ add(dest0, dest0, tmp1);       // dest0 is full
7021 
7022     __ srli(dest1, tmp2, 12);         // 14-bit in dest1
7023 
7024     __ ld(tmp1, Address(src, 3 * sizeof(jlong)));
7025     __ slli(tmp1, tmp1, 14);
7026     __ add(dest1, dest1, tmp1);       // 40-bit in dest1
7027 
7028     __ ld(tmp1, Address(src, 4 * sizeof(jlong)));
7029     __ slli(tmp2, tmp1, 40);
7030     __ add(dest1, dest1, tmp2);       // dest1 is full
7031 
7032     if (dest2->is_valid()) {
7033       __ srli(tmp1, tmp1, 24);
7034       __ mv(dest2, tmp1);               // 2 bits in dest2
7035     } else {
7036 #ifdef ASSERT
7037       Label OK;
7038       __ srli(tmp1, tmp1, 24);
7039       __ beq(zr, tmp1, OK);           // 2 bits
7040       __ stop("high bits of Poly1305 integer should be zero");
7041       __ should_not_reach_here();
7042       __ bind(OK);
7043 #endif
7044     }
7045   }
7046 
7047   // As above, but return only a 128-bit integer, packed into two
7048   // 64-bit registers.
7049   void poly1305_pack_26(Register dest0, Register dest1, Register src, Register tmp1, Register tmp2) {
7050     poly1305_pack_26(dest0, dest1, noreg, src, tmp1, tmp2);
7051   }
7052 
7053   // U_2:U_1:U_0: += (U_2 >> 2) * 5
7054   void poly1305_reduce(Register U_2, Register U_1, Register U_0, Register tmp1, Register tmp2) {
7055     assert_different_registers(U_2, U_1, U_0, tmp1, tmp2);
7056 
7057     // First, U_2:U_1:U_0 += (U_2 >> 2)
7058     __ srli(tmp1, U_2, 2);
7059     __ cad(U_0, U_0, tmp1, tmp2); // Add tmp1 to U_0 with carry output to tmp2
7060     __ andi(U_2, U_2, right_2_bits); // Clear U_2 except for the lowest two bits
7061     __ cad(U_1, U_1, tmp2, tmp2); // Add carry to U_1 with carry output to tmp2
7062     __ add(U_2, U_2, tmp2);
7063 
7064     // Second, U_2:U_1:U_0 += (U_2 >> 2) << 2
7065     __ slli(tmp1, tmp1, 2);
7066     __ cad(U_0, U_0, tmp1, tmp2); // Add tmp1 to U_0 with carry output to tmp2
7067     __ cad(U_1, U_1, tmp2, tmp2); // Add carry to U_1 with carry output to tmp2
7068     __ add(U_2, U_2, tmp2);
7069   }
7070 
7071   // Poly1305, RFC 7539
7072   // void com.sun.crypto.provider.Poly1305.processMultipleBlocks(byte[] input, int offset, int length, long[] aLimbs, long[] rLimbs)
7073 
7074   // Arguments:
7075   //    c_rarg0:   input_start -- where the input is stored
7076   //    c_rarg1:   length
7077   //    c_rarg2:   acc_start -- where the output will be stored
7078   //    c_rarg3:   r_start -- where the randomly generated 128-bit key is stored
7079 
7080   // See https://loup-vaillant.fr/tutorials/poly1305-design for a
7081   // description of the tricks used to simplify and accelerate this
7082   // computation.
7083 
7084   address generate_poly1305_processBlocks() {
7085     __ align(CodeEntryAlignment);
7086     StubId stub_id = StubId::stubgen_poly1305_processBlocks_id;
7087     StubCodeMark mark(this, stub_id);
7088     address start = __ pc();
7089     __ enter();
7090     Label here;
7091 
7092     RegSet saved_regs = RegSet::range(x18, x21);
7093     RegSetIterator<Register> regs = (RegSet::range(x14, x31) - RegSet::range(x22, x27)).begin();
7094     __ push_reg(saved_regs, sp);
7095 
7096     // Arguments
7097     const Register input_start = c_rarg0, length = c_rarg1, acc_start = c_rarg2, r_start = c_rarg3;
7098 
7099     // R_n is the 128-bit randomly-generated key, packed into two
7100     // registers. The caller passes this key to us as long[5], with
7101     // BITS_PER_LIMB = 26.
7102     const Register R_0 = *regs, R_1 = *++regs;
7103     poly1305_pack_26(R_0, R_1, r_start, t1, t2);
7104 
7105     // RR_n is (R_n >> 2) * 5
7106     const Register RR_0 = *++regs, RR_1 = *++regs;
7107     __ srli(t1, R_0, 2);
7108     __ shadd(RR_0, t1, t1, t2, 2);
7109     __ srli(t1, R_1, 2);
7110     __ shadd(RR_1, t1, t1, t2, 2);
7111 
7112     // U_n is the current checksum
7113     const Register U_0 = *++regs, U_1 = *++regs, U_2 = *++regs;
7114     poly1305_pack_26(U_0, U_1, U_2, acc_start, t1, t2);
7115 
7116     static constexpr int BLOCK_LENGTH = 16;
7117     Label DONE, LOOP;
7118 
7119     __ mv(t1, BLOCK_LENGTH);
7120     __ blt(length, t1, DONE); {
7121       __ bind(LOOP);
7122 
7123       // S_n is to be the sum of U_n and the next block of data
7124       const Register S_0 = *++regs, S_1 = *++regs, S_2 = *++regs;
7125       __ ld(S_0, Address(input_start, 0));
7126       __ ld(S_1, Address(input_start, wordSize));
7127 
7128       __ cad(S_0, S_0, U_0, t1); // Add U_0 to S_0 with carry output to t1
7129       __ cadc(S_1, S_1, U_1, t1); // Add U_1 with carry to S_1 with carry output to t1
7130       __ add(S_2, U_2, t1);
7131 
7132       __ addi(S_2, S_2, 1);
7133 
7134       const Register U_0HI = *++regs, U_1HI = *++regs;
7135 
7136       // NB: this logic depends on some of the special properties of
7137       // Poly1305 keys. In particular, because we know that the top
7138       // four bits of R_0 and R_1 are zero, we can add together
7139       // partial products without any risk of needing to propagate a
7140       // carry out.
7141       __ wide_mul(U_0, U_0HI, S_0, R_0);
7142       __ wide_madd(U_0, U_0HI, S_1, RR_1, t1, t2);
7143       __ wide_madd(U_0, U_0HI, S_2, RR_0, t1, t2);
7144 
7145       __ wide_mul(U_1, U_1HI, S_0, R_1);
7146       __ wide_madd(U_1, U_1HI, S_1, R_0, t1, t2);
7147       __ wide_madd(U_1, U_1HI, S_2, RR_1, t1, t2);
7148 
7149       __ andi(U_2, R_0, right_2_bits);
7150       __ mul(U_2, S_2, U_2);
7151 
7152       // Partial reduction mod 2**130 - 5
7153       __ cad(U_1, U_1, U_0HI, t1); // Add U_0HI to U_1 with carry output to t1
7154       __ adc(U_2, U_2, U_1HI, t1);
7155       // Sum is now in U_2:U_1:U_0.
7156 
7157       // U_2:U_1:U_0: += (U_2 >> 2) * 5
7158       poly1305_reduce(U_2, U_1, U_0, t1, t2);
7159 
7160       __ subi(length, length, BLOCK_LENGTH);
7161       __ addi(input_start, input_start, BLOCK_LENGTH);
7162       __ mv(t1, BLOCK_LENGTH);
7163       __ bge(length, t1, LOOP);
7164     }
7165 
7166     // Further reduce modulo 2^130 - 5
7167     poly1305_reduce(U_2, U_1, U_0, t1, t2);
7168 
7169     // Unpack the sum into five 26-bit limbs and write to memory.
7170     // First 26 bits is the first limb
7171     __ slli(t1, U_0, 38); // Take lowest 26 bits
7172     __ srli(t1, t1, 38);
7173     __ sd(t1, Address(acc_start)); // First 26-bit limb
7174 
7175     // 27-52 bits of U_0 is the second limb
7176     __ slli(t1, U_0, 12); // Take next 27-52 bits
7177     __ srli(t1, t1, 38);
7178     __ sd(t1, Address(acc_start, sizeof (jlong))); // Second 26-bit limb
7179 
7180     // Getting 53-64 bits of U_0 and 1-14 bits of U_1 in one register
7181     __ srli(t1, U_0, 52);
7182     __ slli(t2, U_1, 50);
7183     __ srli(t2, t2, 38);
7184     __ add(t1, t1, t2);
7185     __ sd(t1, Address(acc_start, 2 * sizeof (jlong))); // Third 26-bit limb
7186 
7187     // Storing 15-40 bits of U_1
7188     __ slli(t1, U_1, 24); // Already used up 14 bits
7189     __ srli(t1, t1, 38); // Clear all other bits from t1
7190     __ sd(t1, Address(acc_start, 3 * sizeof (jlong))); // Fourth 26-bit limb
7191 
7192     // Storing 41-64 bits of U_1 and first three bits from U_2 in one register
7193     __ srli(t1, U_1, 40);
7194     __ andi(t2, U_2, right_3_bits);
7195     __ slli(t2, t2, 24);
7196     __ add(t1, t1, t2);
7197     __ sd(t1, Address(acc_start, 4 * sizeof (jlong))); // Fifth 26-bit limb
7198 
7199     __ bind(DONE);
7200     __ pop_reg(saved_regs, sp);
7201     __ leave(); // Required for proper stackwalking
7202     __ ret();
7203 
7204     return start;
7205   }
7206 
7207   address generate_arrays_hashcode_powers_of_31() {
7208     assert(UseRVV, "sanity");
7209     const int lmul = 2;
7210     const int stride = MaxVectorSize / sizeof(jint) * lmul;
7211     __ align(CodeEntryAlignment);
7212     StubCodeMark mark(this, "StubRoutines", "arrays_hashcode_powers_of_31");
7213     address start = __ pc();
7214     for (int i = stride; i >= 0; i--) {
7215         jint power_of_31 = 1;
7216         for (int j = i; j > 0; j--) {
7217           power_of_31 = java_multiply(power_of_31, 31);
7218         }
7219         __ emit_int32(power_of_31);
7220     }
7221 
7222     return start;
7223   }
7224 
7225 #endif // COMPILER2
7226 
7227   /**
7228    *  Arguments:
7229    *
7230    * Inputs:
7231    *   c_rarg0   - int crc
7232    *   c_rarg1   - byte* buf
7233    *   c_rarg2   - int length
7234    *
7235    * Output:
7236    *   c_rarg0   - int crc result
7237    */
7238   address generate_updateBytesCRC32() {
7239     assert(UseCRC32Intrinsics, "what are we doing here?");
7240 
7241     __ align(CodeEntryAlignment);
7242     StubId stub_id = StubId::stubgen_updateBytesCRC32_id;
7243     StubCodeMark mark(this, stub_id);
7244 
7245     address start = __ pc();
7246 
7247     // input parameters
7248     const Register crc    = c_rarg0;  // crc
7249     const Register buf    = c_rarg1;  // source java byte array address
7250     const Register len    = c_rarg2;  // length
7251 
7252     BLOCK_COMMENT("Entry:");
7253     __ enter(); // required for proper stackwalking of RuntimeStub frame
7254 
7255     __ kernel_crc32(crc, buf, len,
7256                     c_rarg3, c_rarg4, c_rarg5, c_rarg6, // tmp's for tables
7257                     c_rarg7, t2, t3, t4, t5, t6);       // misc tmps
7258 
7259     __ leave(); // required for proper stackwalking of RuntimeStub frame
7260     __ ret();
7261 
7262     return start;
7263   }
7264 
7265   // exception handler for upcall stubs
7266   address generate_upcall_stub_exception_handler() {
7267     StubId stub_id = StubId::stubgen_upcall_stub_exception_handler_id;
7268     StubCodeMark mark(this, stub_id);
7269     address start = __ pc();
7270 
7271     // Native caller has no idea how to handle exceptions,
7272     // so we just crash here. Up to callee to catch exceptions.
7273     __ verify_oop(x10); // return a exception oop in a0
7274     __ rt_call(CAST_FROM_FN_PTR(address, UpcallLinker::handle_uncaught_exception));
7275     __ should_not_reach_here();
7276 
7277     return start;
7278   }
7279 
7280   // load Method* target of MethodHandle
7281   // j_rarg0 = jobject receiver
7282   // xmethod = Method* result
7283   address generate_upcall_stub_load_target() {
7284 
7285     StubId stub_id = StubId::stubgen_upcall_stub_load_target_id;
7286     StubCodeMark mark(this, stub_id);
7287     address start = __ pc();
7288 
7289     __ resolve_global_jobject(j_rarg0, t0, t1);
7290       // Load target method from receiver
7291     __ load_heap_oop(xmethod, Address(j_rarg0, java_lang_invoke_MethodHandle::form_offset()), t0, t1);
7292     __ load_heap_oop(xmethod, Address(xmethod, java_lang_invoke_LambdaForm::vmentry_offset()), t0, t1);
7293     __ load_heap_oop(xmethod, Address(xmethod, java_lang_invoke_MemberName::method_offset()), t0, t1);
7294     __ access_load_at(T_ADDRESS, IN_HEAP, xmethod,
7295                       Address(xmethod, java_lang_invoke_ResolvedMethodName::vmtarget_offset()),
7296                       noreg, noreg);
7297     __ sd(xmethod, Address(xthread, JavaThread::callee_target_offset())); // just in case callee is deoptimized
7298 
7299     __ ret();
7300 
7301     return start;
7302   }
7303 
7304 #undef __
7305 
7306   // Initialization
7307   void generate_preuniverse_stubs() {
7308     // preuniverse stubs are not needed for riscv
7309   }
7310 
7311   void generate_initial_stubs() {
7312     // Generate initial stubs and initializes the entry points
7313 
7314     // entry points that exist in all platforms Note: This is code
7315     // that could be shared among different platforms - however the
7316     // benefit seems to be smaller than the disadvantage of having a
7317     // much more complicated generator structure. See also comment in
7318     // stubRoutines.hpp.
7319 
7320     StubRoutines::_forward_exception_entry = generate_forward_exception();
7321 
7322     if (UnsafeMemoryAccess::_table == nullptr) {
7323       UnsafeMemoryAccess::create_table(8 + 4); // 8 for copyMemory; 4 for setMemory
7324     }
7325 
7326     StubRoutines::_call_stub_entry =
7327       generate_call_stub(StubRoutines::_call_stub_return_address);
7328 
7329     // is referenced by megamorphic call
7330     StubRoutines::_catch_exception_entry = generate_catch_exception();
7331 
7332     if (UseCRC32Intrinsics) {
7333       StubRoutines::_updateBytesCRC32 = generate_updateBytesCRC32();
7334     }
7335 
7336     if (vmIntrinsics::is_intrinsic_available(vmIntrinsics::_float16ToFloat) &&
7337         vmIntrinsics::is_intrinsic_available(vmIntrinsics::_floatToFloat16)) {
7338       StubRoutines::_hf2f = generate_float16ToFloat();
7339       StubRoutines::_f2hf = generate_floatToFloat16();
7340     }
7341   }
7342 
7343   void generate_continuation_stubs() {
7344     // Continuation stubs:
7345     StubRoutines::_cont_thaw             = generate_cont_thaw();
7346     StubRoutines::_cont_returnBarrier    = generate_cont_returnBarrier();
7347     StubRoutines::_cont_returnBarrierExc = generate_cont_returnBarrier_exception();
7348     StubRoutines::_cont_preempt_stub     = generate_cont_preempt_stub();
7349   }
7350 
7351   void generate_final_stubs() {
7352     // support for verify_oop (must happen after universe_init)
7353     if (VerifyOops) {
7354       StubRoutines::_verify_oop_subroutine_entry = generate_verify_oop();
7355     }
7356 
7357     // arraycopy stubs used by compilers
7358     generate_arraycopy_stubs();
7359 
7360     StubRoutines::_method_entry_barrier = generate_method_entry_barrier();
7361 
7362 #ifdef COMPILER2
7363     if (UseSecondarySupersTable) {
7364       StubRoutines::_lookup_secondary_supers_table_slow_path_stub = generate_lookup_secondary_supers_table_slow_path_stub();
7365       if (!InlineSecondarySupersTest) {
7366         generate_lookup_secondary_supers_table_stub();
7367       }
7368     }
7369 #endif // COMPILER2
7370 
7371     StubRoutines::_upcall_stub_exception_handler = generate_upcall_stub_exception_handler();
7372     StubRoutines::_upcall_stub_load_target = generate_upcall_stub_load_target();
7373 
7374     StubRoutines::riscv::set_completed();
7375   }
7376 
7377   void generate_compiler_stubs() {
7378 #ifdef COMPILER2
7379     if (UseMulAddIntrinsic) {
7380       StubRoutines::_mulAdd = generate_mulAdd();
7381     }
7382 
7383     if (UseMultiplyToLenIntrinsic) {
7384       StubRoutines::_multiplyToLen = generate_multiplyToLen();
7385     }
7386 
7387     if (UseSquareToLenIntrinsic) {
7388       StubRoutines::_squareToLen = generate_squareToLen();
7389     }
7390 
7391     if (UseMontgomeryMultiplyIntrinsic) {
7392       StubId stub_id = StubId::stubgen_montgomeryMultiply_id;
7393       StubCodeMark mark(this, stub_id);
7394       MontgomeryMultiplyGenerator g(_masm, /*squaring*/false);
7395       StubRoutines::_montgomeryMultiply = g.generate_multiply();
7396     }
7397 
7398     if (UseMontgomerySquareIntrinsic) {
7399       StubId stub_id = StubId::stubgen_montgomerySquare_id;
7400       StubCodeMark mark(this, stub_id);
7401       MontgomeryMultiplyGenerator g(_masm, /*squaring*/true);
7402       StubRoutines::_montgomerySquare = g.generate_square();
7403     }
7404 
7405     if (UseAESIntrinsics) {
7406       StubRoutines::_aescrypt_encryptBlock = generate_aescrypt_encryptBlock();
7407       StubRoutines::_aescrypt_decryptBlock = generate_aescrypt_decryptBlock();
7408       StubRoutines::_cipherBlockChaining_encryptAESCrypt = generate_cipherBlockChaining_encryptAESCrypt();
7409       StubRoutines::_cipherBlockChaining_decryptAESCrypt = generate_cipherBlockChaining_decryptAESCrypt();
7410     }
7411 
7412     if (UseAESCTRIntrinsics) {
7413       StubRoutines::_counterMode_AESCrypt = generate_counterMode_AESCrypt();
7414     }
7415 
7416     if (UseGHASHIntrinsics) {
7417       StubRoutines::_ghash_processBlocks = generate_ghash_processBlocks();
7418     }
7419 
7420     if (UseAESCTRIntrinsics && UseGHASHIntrinsics) {
7421       StubRoutines::_galoisCounterMode_AESCrypt = generate_galoisCounterMode_AESCrypt();
7422     }
7423 
7424     if (UsePoly1305Intrinsics) {
7425       StubRoutines::_poly1305_processBlocks = generate_poly1305_processBlocks();
7426     }
7427 
7428     if (UseRVV) {
7429       StubRoutines::_bigIntegerLeftShiftWorker = generate_bigIntegerLeftShift();
7430       StubRoutines::_bigIntegerRightShiftWorker = generate_bigIntegerRightShift();
7431     }
7432 
7433     if (UseVectorizedHashCodeIntrinsic && UseRVV) {
7434       StubRoutines::riscv::_arrays_hashcode_powers_of_31 = generate_arrays_hashcode_powers_of_31();
7435     }
7436 
7437     if (UseSHA256Intrinsics) {
7438       Sha2Generator sha2(_masm, this);
7439       StubRoutines::_sha256_implCompress   = sha2.generate_sha256_implCompress(StubId::stubgen_sha256_implCompress_id);
7440       StubRoutines::_sha256_implCompressMB = sha2.generate_sha256_implCompress(StubId::stubgen_sha256_implCompressMB_id);
7441     }
7442 
7443     if (UseSHA512Intrinsics) {
7444       Sha2Generator sha2(_masm, this);
7445       StubRoutines::_sha512_implCompress   = sha2.generate_sha512_implCompress(StubId::stubgen_sha512_implCompress_id);
7446       StubRoutines::_sha512_implCompressMB = sha2.generate_sha512_implCompress(StubId::stubgen_sha512_implCompressMB_id);
7447     }
7448 
7449     if (UseMD5Intrinsics) {
7450       StubRoutines::_md5_implCompress   = generate_md5_implCompress(StubId::stubgen_md5_implCompress_id);
7451       StubRoutines::_md5_implCompressMB = generate_md5_implCompress(StubId::stubgen_md5_implCompressMB_id);
7452     }
7453 
7454     if (UseChaCha20Intrinsics) {
7455       StubRoutines::_chacha20Block = generate_chacha20Block();
7456     }
7457 
7458     if (UseSHA1Intrinsics) {
7459       StubRoutines::_sha1_implCompress     = generate_sha1_implCompress(StubId::stubgen_sha1_implCompress_id);
7460       StubRoutines::_sha1_implCompressMB   = generate_sha1_implCompress(StubId::stubgen_sha1_implCompressMB_id);
7461     }
7462 
7463     if (UseBASE64Intrinsics) {
7464       StubRoutines::_base64_encodeBlock = generate_base64_encodeBlock();
7465       StubRoutines::_base64_decodeBlock = generate_base64_decodeBlock();
7466     }
7467 
7468     if (UseAdler32Intrinsics) {
7469       StubRoutines::_updateBytesAdler32 = generate_updateBytesAdler32();
7470     }
7471 
7472     generate_compare_long_strings();
7473 
7474     generate_string_indexof_stubs();
7475 
7476 #endif // COMPILER2
7477   }
7478 
7479  public:
7480   StubGenerator(CodeBuffer* code, BlobId blob_id, AOTStubData* stub_data) : StubCodeGenerator(code, blob_id, stub_data) {
7481     switch(blob_id) {
7482     case BlobId::stubgen_preuniverse_id:
7483       generate_preuniverse_stubs();
7484       break;
7485     case BlobId::stubgen_initial_id:
7486       generate_initial_stubs();
7487       break;
7488     case BlobId::stubgen_continuation_id:
7489       generate_continuation_stubs();
7490       break;
7491     case BlobId::stubgen_compiler_id:
7492       generate_compiler_stubs();
7493       break;
7494     case BlobId::stubgen_final_id:
7495       generate_final_stubs();
7496       break;
7497     default:
7498       fatal("unexpected blob id: %s", StubInfo::name(blob_id));
7499       break;
7500     };
7501   }
7502 }; // end class declaration
7503 
7504 void StubGenerator_generate(CodeBuffer* code, BlobId blob_id, AOTStubData* stub_data) {
7505   StubGenerator g(code, blob_id, stub_data);
7506 }