1 /*
2 * Copyright (c) 2003, 2026, Oracle and/or its affiliates. All rights reserved.
3 * Copyright (c) 2014, 2025, Red Hat Inc. All rights reserved.
4 * Copyright (c) 2020, 2025, Huawei Technologies Co., Ltd. All rights reserved.
5 * DO NOT ALTER OR REMOVE COPYRIGHT NOTICES OR THIS FILE HEADER.
6 *
7 * This code is free software; you can redistribute it and/or modify it
8 * under the terms of the GNU General Public License version 2 only, as
9 * published by the Free Software Foundation.
10 *
11 * This code is distributed in the hope that it will be useful, but WITHOUT
12 * ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
13 * FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License
14 * version 2 for more details (a copy is included in the LICENSE file that
15 * accompanied this code).
16 *
17 * You should have received a copy of the GNU General Public License version
18 * 2 along with this work; if not, write to the Free Software Foundation,
19 * Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA.
20 *
21 * Please contact Oracle, 500 Oracle Parkway, Redwood Shores, CA 94065 USA
22 * or visit www.oracle.com if you need additional information or have any
23 * questions.
24 *
25 */
26
27 #include "asm/macroAssembler.hpp"
28 #include "asm/macroAssembler.inline.hpp"
29 #include "compiler/oopMap.hpp"
30 #include "gc/shared/barrierSet.hpp"
31 #include "gc/shared/barrierSetAssembler.hpp"
32 #include "interpreter/interpreter.hpp"
33 #include "memory/universe.hpp"
34 #include "nativeInst_riscv.hpp"
35 #include "oops/instanceOop.hpp"
36 #include "oops/method.hpp"
37 #include "oops/objArrayKlass.hpp"
38 #include "oops/oop.inline.hpp"
39 #include "prims/methodHandles.hpp"
40 #include "prims/upcallLinker.hpp"
41 #include "runtime/continuation.hpp"
42 #include "runtime/continuationEntry.inline.hpp"
43 #include "runtime/frame.inline.hpp"
44 #include "runtime/handles.inline.hpp"
45 #include "runtime/javaThread.hpp"
46 #include "runtime/sharedRuntime.hpp"
47 #include "runtime/stubCodeGenerator.hpp"
48 #include "runtime/stubRoutines.hpp"
49 #include "utilities/align.hpp"
50 #include "utilities/powerOfTwo.hpp"
51 #ifdef COMPILER2
52 #include "opto/runtime.hpp"
53 #endif
54
55 // Declaration and definition of StubGenerator (no .hpp file).
56 // For a more detailed description of the stub routine structure
57 // see the comment in stubRoutines.hpp
58
59 #undef __
60 #define __ _masm->
61
62 #ifdef PRODUCT
63 #define BLOCK_COMMENT(str) /* nothing */
64 #else
65 #define BLOCK_COMMENT(str) __ block_comment(str)
66 #endif
67
68 #define BIND(label) bind(label); BLOCK_COMMENT(#label ":")
69
70 // Stub Code definitions
71
72 class StubGenerator: public StubCodeGenerator {
73 private:
74
75 #ifdef PRODUCT
76 #define inc_counter_np(counter) ((void)0)
77 #else
78 void inc_counter_np_(uint& counter) {
79 __ incrementw(ExternalAddress((address)&counter));
80 }
81 #define inc_counter_np(counter) \
82 BLOCK_COMMENT("inc_counter " #counter); \
83 inc_counter_np_(counter);
84 #endif
85
86 // Call stubs are used to call Java from C
87 //
88 // Arguments:
89 // c_rarg0: call wrapper address address
90 // c_rarg1: result address
91 // c_rarg2: result type BasicType
92 // c_rarg3: method Method*
93 // c_rarg4: (interpreter) entry point address
94 // c_rarg5: parameters intptr_t*
95 // c_rarg6: parameter size (in words) int
96 // c_rarg7: thread Thread*
97 //
98 // There is no return from the stub itself as any Java result
99 // is written to result
100 //
101 // we save x1 (ra) as the return PC at the base of the frame and
102 // link x8 (fp) below it as the frame pointer installing sp (x2)
103 // into fp.
104 //
105 // we save x10-x17, which accounts for all the c arguments.
106 //
107 // TODO: strictly do we need to save them all? they are treated as
108 // volatile by C so could we omit saving the ones we are going to
109 // place in global registers (thread? method?) or those we only use
110 // during setup of the Java call?
111 //
112 // we don't need to save x5 which C uses as an indirect result location
113 // return register.
114 //
115 // we don't need to save x6-x7 and x28-x31 which both C and Java treat as
116 // volatile
117 //
118 // we save x9, x18-x27, f8-f9, and f18-f27 which Java uses as temporary
119 // registers and C expects to be callee-save
120 //
121 // so the stub frame looks like this when we enter Java code
122 //
123 // [ return_from_Java ] <--- sp
124 // [ argument word n ]
125 // ...
126 // -35 [ argument word 1 ]
127 // -34 [ saved FRM in Floating-point Control and Status Register ] <--- sp_after_call
128 // -33 [ saved f27 ]
129 // -32 [ saved f26 ]
130 // -31 [ saved f25 ]
131 // -30 [ saved f24 ]
132 // -29 [ saved f23 ]
133 // -28 [ saved f22 ]
134 // -27 [ saved f21 ]
135 // -26 [ saved f20 ]
136 // -25 [ saved f19 ]
137 // -24 [ saved f18 ]
138 // -23 [ saved f9 ]
139 // -22 [ saved f8 ]
140 // -21 [ saved x27 ]
141 // -20 [ saved x26 ]
142 // -19 [ saved x25 ]
143 // -18 [ saved x24 ]
144 // -17 [ saved x23 ]
145 // -16 [ saved x22 ]
146 // -15 [ saved x21 ]
147 // -14 [ saved x20 ]
148 // -13 [ saved x19 ]
149 // -12 [ saved x18 ]
150 // -11 [ saved x9 ]
151 // -10 [ call wrapper (x10) ]
152 // -9 [ result (x11) ]
153 // -8 [ result type (x12) ]
154 // -7 [ method (x13) ]
155 // -6 [ entry point (x14) ]
156 // -5 [ parameters (x15) ]
157 // -4 [ parameter size (x16) ]
158 // -3 [ thread (x17) ]
159 // -2 [ saved fp (x8) ]
160 // -1 [ saved ra (x1) ]
161 // 0 [ ] <--- fp == saved sp (x2)
162
163 // Call stub stack layout word offsets from fp
164 enum call_stub_layout {
165 sp_after_call_off = -34,
166
167 frm_off = sp_after_call_off,
168 f27_off = -33,
169 f26_off = -32,
170 f25_off = -31,
171 f24_off = -30,
172 f23_off = -29,
173 f22_off = -28,
174 f21_off = -27,
175 f20_off = -26,
176 f19_off = -25,
177 f18_off = -24,
178 f9_off = -23,
179 f8_off = -22,
180
181 x27_off = -21,
182 x26_off = -20,
183 x25_off = -19,
184 x24_off = -18,
185 x23_off = -17,
186 x22_off = -16,
187 x21_off = -15,
188 x20_off = -14,
189 x19_off = -13,
190 x18_off = -12,
191 x9_off = -11,
192
193 call_wrapper_off = -10,
194 result_off = -9,
195 result_type_off = -8,
196 method_off = -7,
197 entry_point_off = -6,
198 parameters_off = -5,
199 parameter_size_off = -4,
200 thread_off = -3,
201 fp_f = -2,
202 retaddr_off = -1,
203 };
204
205 address generate_call_stub(address& return_address) {
206 assert((int)frame::entry_frame_after_call_words == -(int)sp_after_call_off + 1 &&
207 (int)frame::entry_frame_call_wrapper_offset == (int)call_wrapper_off,
208 "adjust this code");
209
210 StubId stub_id = StubId::stubgen_call_stub_id;
211 StubCodeMark mark(this, stub_id);
212 address start = __ pc();
213
214 const Address sp_after_call (fp, sp_after_call_off * wordSize);
215
216 const Address frm_save (fp, frm_off * wordSize);
217 const Address call_wrapper (fp, call_wrapper_off * wordSize);
218 const Address result (fp, result_off * wordSize);
219 const Address result_type (fp, result_type_off * wordSize);
220 const Address method (fp, method_off * wordSize);
221 const Address entry_point (fp, entry_point_off * wordSize);
222 const Address parameters (fp, parameters_off * wordSize);
223 const Address parameter_size(fp, parameter_size_off * wordSize);
224
225 const Address thread (fp, thread_off * wordSize);
226
227 const Address f27_save (fp, f27_off * wordSize);
228 const Address f26_save (fp, f26_off * wordSize);
229 const Address f25_save (fp, f25_off * wordSize);
230 const Address f24_save (fp, f24_off * wordSize);
231 const Address f23_save (fp, f23_off * wordSize);
232 const Address f22_save (fp, f22_off * wordSize);
233 const Address f21_save (fp, f21_off * wordSize);
234 const Address f20_save (fp, f20_off * wordSize);
235 const Address f19_save (fp, f19_off * wordSize);
236 const Address f18_save (fp, f18_off * wordSize);
237 const Address f9_save (fp, f9_off * wordSize);
238 const Address f8_save (fp, f8_off * wordSize);
239
240 const Address x27_save (fp, x27_off * wordSize);
241 const Address x26_save (fp, x26_off * wordSize);
242 const Address x25_save (fp, x25_off * wordSize);
243 const Address x24_save (fp, x24_off * wordSize);
244 const Address x23_save (fp, x23_off * wordSize);
245 const Address x22_save (fp, x22_off * wordSize);
246 const Address x21_save (fp, x21_off * wordSize);
247 const Address x20_save (fp, x20_off * wordSize);
248 const Address x19_save (fp, x19_off * wordSize);
249 const Address x18_save (fp, x18_off * wordSize);
250
251 const Address x9_save (fp, x9_off * wordSize);
252
253 // stub code
254
255 address riscv_entry = __ pc();
256
257 // set up frame and move sp to end of save area
258 __ enter();
259 __ addi(sp, fp, sp_after_call_off * wordSize);
260
261 // save register parameters and Java temporary/global registers
262 // n.b. we save thread even though it gets installed in
263 // xthread because we want to sanity check tp later
264 __ sd(c_rarg7, thread);
265 __ sw(c_rarg6, parameter_size);
266 __ sd(c_rarg5, parameters);
267 __ sd(c_rarg4, entry_point);
268 __ sd(c_rarg3, method);
269 __ sd(c_rarg2, result_type);
270 __ sd(c_rarg1, result);
271 __ sd(c_rarg0, call_wrapper);
272
273 __ sd(x9, x9_save);
274
275 __ sd(x18, x18_save);
276 __ sd(x19, x19_save);
277 __ sd(x20, x20_save);
278 __ sd(x21, x21_save);
279 __ sd(x22, x22_save);
280 __ sd(x23, x23_save);
281 __ sd(x24, x24_save);
282 __ sd(x25, x25_save);
283 __ sd(x26, x26_save);
284 __ sd(x27, x27_save);
285
286 __ fsd(f8, f8_save);
287 __ fsd(f9, f9_save);
288 __ fsd(f18, f18_save);
289 __ fsd(f19, f19_save);
290 __ fsd(f20, f20_save);
291 __ fsd(f21, f21_save);
292 __ fsd(f22, f22_save);
293 __ fsd(f23, f23_save);
294 __ fsd(f24, f24_save);
295 __ fsd(f25, f25_save);
296 __ fsd(f26, f26_save);
297 __ fsd(f27, f27_save);
298
299 __ frrm(t0);
300 __ sd(t0, frm_save);
301 // Set frm to the state we need. We do want Round to Nearest. We
302 // don't want non-IEEE rounding modes.
303 Label skip_fsrmi;
304 guarantee(__ RoundingMode::rne == 0, "must be");
305 __ beqz(t0, skip_fsrmi);
306 __ fsrmi(__ RoundingMode::rne);
307 __ bind(skip_fsrmi);
308
309 // install Java thread in global register now we have saved
310 // whatever value it held
311 __ mv(xthread, c_rarg7);
312
313 // And method
314 __ mv(xmethod, c_rarg3);
315
316 // set up the heapbase register
317 __ reinit_heapbase();
318
319 #ifdef ASSERT
320 // make sure we have no pending exceptions
321 {
322 Label L;
323 __ ld(t0, Address(xthread, in_bytes(Thread::pending_exception_offset())));
324 __ beqz(t0, L);
325 __ stop("StubRoutines::call_stub: entered with pending exception");
326 __ BIND(L);
327 }
328 #endif
329 // pass parameters if any
330 __ mv(esp, sp);
331 __ slli(t0, c_rarg6, LogBytesPerWord);
332 __ sub(t0, sp, t0); // Move SP out of the way
333 __ andi(sp, t0, -2 * wordSize);
334
335 BLOCK_COMMENT("pass parameters if any");
336 Label parameters_done;
337 // parameter count is still in c_rarg6
338 // and parameter pointer identifying param 1 is in c_rarg5
339 __ beqz(c_rarg6, parameters_done);
340
341 address loop = __ pc();
342 __ ld(t0, Address(c_rarg5, 0));
343 __ addi(c_rarg5, c_rarg5, wordSize);
344 __ subi(c_rarg6, c_rarg6, 1);
345 __ push_reg(t0);
346 __ bgtz(c_rarg6, loop);
347
348 __ BIND(parameters_done);
349
350 // call Java entry -- passing methdoOop, and current sp
351 // xmethod: Method*
352 // x19_sender_sp: sender sp
353 BLOCK_COMMENT("call Java function");
354 __ mv(x19_sender_sp, sp);
355 __ jalr(c_rarg4);
356
357 // save current address for use by exception handling code
358
359 return_address = __ pc();
360
361 // store result depending on type (everything that is not
362 // T_OBJECT, T_LONG, T_FLOAT or T_DOUBLE is treated as T_INT)
363 // n.b. this assumes Java returns an integral result in x10
364 // and a floating result in j_farg0
365 __ ld(j_rarg2, result);
366 Label is_long, is_float, is_double, exit;
367 __ ld(j_rarg1, result_type);
368 __ mv(t0, (u1)T_OBJECT);
369 __ beq(j_rarg1, t0, is_long);
370 __ mv(t0, (u1)T_LONG);
371 __ beq(j_rarg1, t0, is_long);
372 __ mv(t0, (u1)T_FLOAT);
373 __ beq(j_rarg1, t0, is_float);
374 __ mv(t0, (u1)T_DOUBLE);
375 __ beq(j_rarg1, t0, is_double);
376
377 // handle T_INT case
378 __ sw(x10, Address(j_rarg2));
379
380 __ BIND(exit);
381
382 // pop parameters
383 __ addi(esp, fp, sp_after_call_off * wordSize);
384
385 #ifdef ASSERT
386 // verify that threads correspond
387 {
388 Label L, S;
389 __ ld(t0, thread);
390 __ bne(xthread, t0, S);
391 __ get_thread(t0);
392 __ beq(xthread, t0, L);
393 __ BIND(S);
394 __ stop("StubRoutines::call_stub: threads must correspond");
395 __ BIND(L);
396 }
397 #endif
398
399 __ pop_cont_fastpath(xthread);
400
401 // restore callee-save registers
402 __ fld(f27, f27_save);
403 __ fld(f26, f26_save);
404 __ fld(f25, f25_save);
405 __ fld(f24, f24_save);
406 __ fld(f23, f23_save);
407 __ fld(f22, f22_save);
408 __ fld(f21, f21_save);
409 __ fld(f20, f20_save);
410 __ fld(f19, f19_save);
411 __ fld(f18, f18_save);
412 __ fld(f9, f9_save);
413 __ fld(f8, f8_save);
414
415 __ ld(x27, x27_save);
416 __ ld(x26, x26_save);
417 __ ld(x25, x25_save);
418 __ ld(x24, x24_save);
419 __ ld(x23, x23_save);
420 __ ld(x22, x22_save);
421 __ ld(x21, x21_save);
422 __ ld(x20, x20_save);
423 __ ld(x19, x19_save);
424 __ ld(x18, x18_save);
425
426 __ ld(x9, x9_save);
427
428 // restore frm
429 Label skip_fsrm;
430 __ ld(t0, frm_save);
431 __ frrm(t1);
432 __ beq(t0, t1, skip_fsrm);
433 __ fsrm(t0);
434 __ bind(skip_fsrm);
435
436 __ ld(c_rarg0, call_wrapper);
437 __ ld(c_rarg1, result);
438 __ ld(c_rarg2, result_type);
439 __ ld(c_rarg3, method);
440 __ ld(c_rarg4, entry_point);
441 __ ld(c_rarg5, parameters);
442 __ ld(c_rarg6, parameter_size);
443 __ ld(c_rarg7, thread);
444
445 // leave frame and return to caller
446 __ leave();
447 __ ret();
448
449 // handle return types different from T_INT
450
451 __ BIND(is_long);
452 __ sd(x10, Address(j_rarg2, 0));
453 __ j(exit);
454
455 __ BIND(is_float);
456 __ fsw(j_farg0, Address(j_rarg2, 0), t0);
457 __ j(exit);
458
459 __ BIND(is_double);
460 __ fsd(j_farg0, Address(j_rarg2, 0), t0);
461 __ j(exit);
462
463 return start;
464 }
465
466 // Return point for a Java call if there's an exception thrown in
467 // Java code. The exception is caught and transformed into a
468 // pending exception stored in JavaThread that can be tested from
469 // within the VM.
470 //
471 // Note: Usually the parameters are removed by the callee. In case
472 // of an exception crossing an activation frame boundary, that is
473 // not the case if the callee is compiled code => need to setup the
474 // sp.
475 //
476 // x10: exception oop
477
478 address generate_catch_exception() {
479 StubId stub_id = StubId::stubgen_catch_exception_id;
480 StubCodeMark mark(this, stub_id);
481 address start = __ pc();
482
483 // same as in generate_call_stub():
484 const Address thread(fp, thread_off * wordSize);
485
486 #ifdef ASSERT
487 // verify that threads correspond
488 {
489 Label L, S;
490 __ ld(t0, thread);
491 __ bne(xthread, t0, S);
492 __ get_thread(t0);
493 __ beq(xthread, t0, L);
494 __ bind(S);
495 __ stop("StubRoutines::catch_exception: threads must correspond");
496 __ bind(L);
497 }
498 #endif
499
500 // set pending exception
501 __ verify_oop(x10);
502
503 __ sd(x10, Address(xthread, Thread::pending_exception_offset()));
504 __ mv(t0, (address)__FILE__);
505 __ sd(t0, Address(xthread, Thread::exception_file_offset()));
506 __ mv(t0, (int)__LINE__);
507 __ sw(t0, Address(xthread, Thread::exception_line_offset()));
508
509 // complete return to VM
510 assert(StubRoutines::_call_stub_return_address != nullptr,
511 "_call_stub_return_address must have been generated before");
512 __ j(RuntimeAddress(StubRoutines::_call_stub_return_address));
513
514 return start;
515 }
516
517 // Continuation point for runtime calls returning with a pending
518 // exception. The pending exception check happened in the runtime
519 // or native call stub. The pending exception in Thread is
520 // converted into a Java-level exception.
521 //
522 // Contract with Java-level exception handlers:
523 // x10: exception
524 // x13: throwing pc
525 //
526 // NOTE: At entry of this stub, exception-pc must be in RA !!
527
528 // NOTE: this is always used as a jump target within generated code
529 // so it just needs to be generated code with no x86 prolog
530
531 address generate_forward_exception() {
532 StubId stub_id = StubId::stubgen_forward_exception_id;
533 StubCodeMark mark(this, stub_id);
534 address start = __ pc();
535
536 // Upon entry, RA points to the return address returning into
537 // Java (interpreted or compiled) code; i.e., the return address
538 // becomes the throwing pc.
539 //
540 // Arguments pushed before the runtime call are still on the stack
541 // but the exception handler will reset the stack pointer ->
542 // ignore them. A potential result in registers can be ignored as
543 // well.
544
545 #ifdef ASSERT
546 // make sure this code is only executed if there is a pending exception
547 {
548 Label L;
549 __ ld(t0, Address(xthread, Thread::pending_exception_offset()));
550 __ bnez(t0, L);
551 __ stop("StubRoutines::forward exception: no pending exception (1)");
552 __ bind(L);
553 }
554 #endif
555
556 // compute exception handler into x9
557
558 // call the VM to find the handler address associated with the
559 // caller address. pass thread in x10 and caller pc (ret address)
560 // in x11. n.b. the caller pc is in ra, unlike x86 where it is on
561 // the stack.
562 __ mv(c_rarg1, ra);
563 // ra will be trashed by the VM call so we move it to x9
564 // (callee-saved) because we also need to pass it to the handler
565 // returned by this call.
566 __ mv(x9, ra);
567 BLOCK_COMMENT("call exception_handler_for_return_address");
568 __ call_VM_leaf(CAST_FROM_FN_PTR(address,
569 SharedRuntime::exception_handler_for_return_address),
570 xthread, c_rarg1);
571 // we should not really care that ra is no longer the callee
572 // address. we saved the value the handler needs in x9 so we can
573 // just copy it to x13. however, the C2 handler will push its own
574 // frame and then calls into the VM and the VM code asserts that
575 // the PC for the frame above the handler belongs to a compiled
576 // Java method. So, we restore ra here to satisfy that assert.
577 __ mv(ra, x9);
578 // setup x10 & x13 & clear pending exception
579 __ mv(x13, x9);
580 __ mv(x9, x10);
581 __ ld(x10, Address(xthread, Thread::pending_exception_offset()));
582 __ sd(zr, Address(xthread, Thread::pending_exception_offset()));
583
584 #ifdef ASSERT
585 // make sure exception is set
586 {
587 Label L;
588 __ bnez(x10, L);
589 __ stop("StubRoutines::forward exception: no pending exception (2)");
590 __ bind(L);
591 }
592 #endif
593
594 // continue at exception handler
595 // x10: exception
596 // x13: throwing pc
597 // x9: exception handler
598 __ verify_oop(x10);
599 __ jr(x9);
600
601 return start;
602 }
603
604 // Non-destructive plausibility checks for oops
605 //
606 // Arguments:
607 // x10: oop to verify
608 // t0: error message
609 //
610 // Stack after saving c_rarg3:
611 // [tos + 0]: saved c_rarg3
612 // [tos + 1]: saved c_rarg2
613 // [tos + 2]: saved ra
614 // [tos + 3]: saved t1
615 // [tos + 4]: saved x10
616 // [tos + 5]: saved t0
617 address generate_verify_oop() {
618
619 StubId stub_id = StubId::stubgen_verify_oop_id;
620 StubCodeMark mark(this, stub_id);
621 address start = __ pc();
622
623 Label exit, error;
624
625 __ push_reg(RegSet::of(c_rarg2, c_rarg3), sp); // save c_rarg2 and c_rarg3
626
627 __ la(c_rarg2, ExternalAddress((address) StubRoutines::verify_oop_count_addr()));
628 __ ld(c_rarg3, Address(c_rarg2));
629 __ addi(c_rarg3, c_rarg3, 1);
630 __ sd(c_rarg3, Address(c_rarg2));
631
632 // object is in x10
633 // make sure object is 'reasonable'
634 __ beqz(x10, exit); // if obj is null it is OK
635
636 BarrierSetAssembler* bs_asm = BarrierSet::barrier_set()->barrier_set_assembler();
637 bs_asm->check_oop(_masm, x10, c_rarg2, c_rarg3, error);
638
639 // return if everything seems ok
640 __ bind(exit);
641
642 __ pop_reg(RegSet::of(c_rarg2, c_rarg3), sp); // pop c_rarg2 and c_rarg3
643 __ ret();
644
645 // handle errors
646 __ bind(error);
647 __ pop_reg(RegSet::of(c_rarg2, c_rarg3), sp); // pop c_rarg2 and c_rarg3
648
649 __ push_reg(RegSet::range(x0, x31), sp);
650 // debug(char* msg, int64_t pc, int64_t regs[])
651 __ mv(c_rarg0, t0); // pass address of error message
652 __ mv(c_rarg1, ra); // pass return address
653 __ mv(c_rarg2, sp); // pass address of regs on stack
654 #ifndef PRODUCT
655 assert(frame::arg_reg_save_area_bytes == 0, "not expecting frame reg save area");
656 #endif
657 BLOCK_COMMENT("call MacroAssembler::debug");
658 __ rt_call(CAST_FROM_FN_PTR(address, MacroAssembler::debug64));
659 __ ebreak();
660
661 return start;
662 }
663
664 // The inner part of zero_words().
665 //
666 // Inputs:
667 // x28: the HeapWord-aligned base address of an array to zero.
668 // x29: the count in HeapWords, x29 > 0.
669 //
670 // Returns x28 and x29, adjusted for the caller to clear.
671 // x28: the base address of the tail of words left to clear.
672 // x29: the number of words in the tail.
673 // x29 < MacroAssembler::zero_words_block_size.
674
675 address generate_zero_blocks() {
676 Label done;
677
678 const Register base = x28, cnt = x29, tmp1 = x30, tmp2 = x31;
679
680 __ align(CodeEntryAlignment);
681 StubId stub_id = StubId::stubgen_zero_blocks_id;
682 StubCodeMark mark(this, stub_id);
683 address start = __ pc();
684
685 if (UseBlockZeroing) {
686 int zicboz_block_size = VM_Version::zicboz_block_size.value();
687 // Ensure count >= 2 * zicboz_block_size so that it still deserves
688 // a cbo.zero after alignment.
689 Label small;
690 int low_limit = MAX2(2 * zicboz_block_size, (int)BlockZeroingLowLimit) / wordSize;
691 __ mv(tmp1, low_limit);
692 __ blt(cnt, tmp1, small);
693 __ zero_dcache_blocks(base, cnt, tmp1, tmp2);
694 __ bind(small);
695 }
696
697 {
698 // Clear the remaining blocks.
699 Label loop;
700 __ mv(tmp1, MacroAssembler::zero_words_block_size);
701 __ blt(cnt, tmp1, done);
702 __ bind(loop);
703 for (int i = 0; i < MacroAssembler::zero_words_block_size; i++) {
704 __ sd(zr, Address(base, i * wordSize));
705 }
706 __ addi(base, base, MacroAssembler::zero_words_block_size * wordSize);
707 __ subi(cnt, cnt, MacroAssembler::zero_words_block_size);
708 __ bge(cnt, tmp1, loop);
709 __ bind(done);
710 }
711
712 __ ret();
713
714 return start;
715 }
716
717 typedef enum {
718 copy_forwards = 1,
719 copy_backwards = -1
720 } copy_direction;
721
722 // Bulk copy of blocks of 8 words.
723 //
724 // count is a count of words.
725 //
726 // Precondition: count >= 8
727 //
728 // Postconditions:
729 //
730 // The least significant bit of count contains the remaining count
731 // of words to copy. The rest of count is trash.
732 //
733 // s and d are adjusted to point to the remaining words to copy
734 //
735 address generate_copy_longs(StubId stub_id, Register s, Register d, Register count) {
736 BasicType type;
737 copy_direction direction;
738 switch (stub_id) {
739 case StubId::stubgen_copy_byte_f_id:
740 direction = copy_forwards;
741 type = T_BYTE;
742 break;
743 case StubId::stubgen_copy_byte_b_id:
744 direction = copy_backwards;
745 type = T_BYTE;
746 break;
747 default:
748 ShouldNotReachHere();
749 }
750 int unit = wordSize * direction;
751 int bias = wordSize;
752
753 const Register tmp_reg0 = x13, tmp_reg1 = x14, tmp_reg2 = x15, tmp_reg3 = x16,
754 tmp_reg4 = x17, tmp_reg5 = x7, tmp_reg6 = x28, tmp_reg7 = x29;
755
756 const Register stride = x30;
757
758 assert_different_registers(t0, tmp_reg0, tmp_reg1, tmp_reg2, tmp_reg3,
759 tmp_reg4, tmp_reg5, tmp_reg6, tmp_reg7);
760 assert_different_registers(s, d, count, t0);
761
762 Label again, drain;
763 StubCodeMark mark(this, stub_id);
764 __ align(CodeEntryAlignment);
765 address start = __ pc();
766
767 if (direction == copy_forwards) {
768 __ sub(s, s, bias);
769 __ sub(d, d, bias);
770 }
771
772 #ifdef ASSERT
773 // Make sure we are never given < 8 words
774 {
775 Label L;
776
777 __ mv(t0, 8);
778 __ bge(count, t0, L);
779 __ stop("genrate_copy_longs called with < 8 words");
780 __ bind(L);
781 }
782 #endif
783
784 __ ld(tmp_reg0, Address(s, 1 * unit));
785 __ ld(tmp_reg1, Address(s, 2 * unit));
786 __ ld(tmp_reg2, Address(s, 3 * unit));
787 __ ld(tmp_reg3, Address(s, 4 * unit));
788 __ ld(tmp_reg4, Address(s, 5 * unit));
789 __ ld(tmp_reg5, Address(s, 6 * unit));
790 __ ld(tmp_reg6, Address(s, 7 * unit));
791 __ ld(tmp_reg7, Address(s, 8 * unit));
792 __ addi(s, s, 8 * unit);
793
794 __ subi(count, count, 16);
795 __ bltz(count, drain);
796
797 __ bind(again);
798
799 __ sd(tmp_reg0, Address(d, 1 * unit));
800 __ sd(tmp_reg1, Address(d, 2 * unit));
801 __ sd(tmp_reg2, Address(d, 3 * unit));
802 __ sd(tmp_reg3, Address(d, 4 * unit));
803 __ sd(tmp_reg4, Address(d, 5 * unit));
804 __ sd(tmp_reg5, Address(d, 6 * unit));
805 __ sd(tmp_reg6, Address(d, 7 * unit));
806 __ sd(tmp_reg7, Address(d, 8 * unit));
807
808 __ ld(tmp_reg0, Address(s, 1 * unit));
809 __ ld(tmp_reg1, Address(s, 2 * unit));
810 __ ld(tmp_reg2, Address(s, 3 * unit));
811 __ ld(tmp_reg3, Address(s, 4 * unit));
812 __ ld(tmp_reg4, Address(s, 5 * unit));
813 __ ld(tmp_reg5, Address(s, 6 * unit));
814 __ ld(tmp_reg6, Address(s, 7 * unit));
815 __ ld(tmp_reg7, Address(s, 8 * unit));
816
817 __ addi(s, s, 8 * unit);
818 __ addi(d, d, 8 * unit);
819
820 __ subi(count, count, 8);
821 __ bgez(count, again);
822
823 // Drain
824 __ bind(drain);
825
826 __ sd(tmp_reg0, Address(d, 1 * unit));
827 __ sd(tmp_reg1, Address(d, 2 * unit));
828 __ sd(tmp_reg2, Address(d, 3 * unit));
829 __ sd(tmp_reg3, Address(d, 4 * unit));
830 __ sd(tmp_reg4, Address(d, 5 * unit));
831 __ sd(tmp_reg5, Address(d, 6 * unit));
832 __ sd(tmp_reg6, Address(d, 7 * unit));
833 __ sd(tmp_reg7, Address(d, 8 * unit));
834 __ addi(d, d, 8 * unit);
835
836 {
837 Label L1, L2;
838 __ test_bit(t0, count, 2);
839 __ beqz(t0, L1);
840
841 __ ld(tmp_reg0, Address(s, 1 * unit));
842 __ ld(tmp_reg1, Address(s, 2 * unit));
843 __ ld(tmp_reg2, Address(s, 3 * unit));
844 __ ld(tmp_reg3, Address(s, 4 * unit));
845 __ addi(s, s, 4 * unit);
846
847 __ sd(tmp_reg0, Address(d, 1 * unit));
848 __ sd(tmp_reg1, Address(d, 2 * unit));
849 __ sd(tmp_reg2, Address(d, 3 * unit));
850 __ sd(tmp_reg3, Address(d, 4 * unit));
851 __ addi(d, d, 4 * unit);
852
853 __ bind(L1);
854
855 if (direction == copy_forwards) {
856 __ addi(s, s, bias);
857 __ addi(d, d, bias);
858 }
859
860 __ test_bit(t0, count, 1);
861 __ beqz(t0, L2);
862 if (direction == copy_backwards) {
863 __ addi(s, s, 2 * unit);
864 __ ld(tmp_reg0, Address(s));
865 __ ld(tmp_reg1, Address(s, wordSize));
866 __ addi(d, d, 2 * unit);
867 __ sd(tmp_reg0, Address(d));
868 __ sd(tmp_reg1, Address(d, wordSize));
869 } else {
870 __ ld(tmp_reg0, Address(s));
871 __ ld(tmp_reg1, Address(s, wordSize));
872 __ addi(s, s, 2 * unit);
873 __ sd(tmp_reg0, Address(d));
874 __ sd(tmp_reg1, Address(d, wordSize));
875 __ addi(d, d, 2 * unit);
876 }
877 __ bind(L2);
878 }
879
880 __ ret();
881
882 return start;
883 }
884
885 typedef void (MacroAssembler::*copy_insn)(Register Rd, const Address &adr, Register temp);
886
887 void copy_memory_v(Register s, Register d, Register count, int step) {
888 bool is_backward = step < 0;
889 int granularity = g_uabs(step);
890
891 const Register src = x30, dst = x31, vl = x14, cnt = x15, tmp1 = x16, tmp2 = x17;
892 assert_different_registers(s, d, cnt, vl, tmp1, tmp2);
893 Assembler::SEW sew = Assembler::elembytes_to_sew(granularity);
894 Label loop_forward, loop_backward, done;
895
896 __ mv(dst, d);
897 __ mv(src, s);
898 __ mv(cnt, count);
899
900 __ bind(loop_forward);
901 __ vsetvli(vl, cnt, sew, Assembler::m8);
902 if (is_backward) {
903 __ bne(vl, cnt, loop_backward);
904 }
905
906 __ vlex_v(v0, src, sew);
907 __ sub(cnt, cnt, vl);
908 if (sew != Assembler::e8) {
909 // when sew == e8 (e.g., elem size is 1 byte), slli R, R, 0 is a nop and unnecessary
910 __ slli(vl, vl, sew);
911 }
912 __ add(src, src, vl);
913
914 __ vsex_v(v0, dst, sew);
915 __ add(dst, dst, vl);
916 __ bnez(cnt, loop_forward);
917
918 if (is_backward) {
919 __ j(done);
920
921 __ bind(loop_backward);
922 __ sub(t0, cnt, vl);
923 if (sew != Assembler::e8) {
924 // when sew == e8 (e.g., elem size is 1 byte), slli R, R, 0 is a nop and unnecessary
925 __ slli(t0, t0, sew);
926 }
927 __ add(tmp1, s, t0);
928 __ vlex_v(v0, tmp1, sew);
929 __ add(tmp2, d, t0);
930 __ vsex_v(v0, tmp2, sew);
931 __ sub(cnt, cnt, vl);
932 __ bnez(cnt, loop_forward);
933 __ bind(done);
934 }
935 }
936
937 // All-singing all-dancing memory copy.
938 //
939 // Copy count units of memory from s to d. The size of a unit is
940 // step, which can be positive or negative depending on the direction
941 // of copy.
942 //
943 void copy_memory(DecoratorSet decorators, BasicType type, bool is_aligned,
944 Register s, Register d, Register count, int step) {
945 BarrierSetAssembler* bs_asm = BarrierSet::barrier_set()->barrier_set_assembler();
946 if (UseRVV && (!is_reference_type(type) || bs_asm->supports_rvv_arraycopy())) {
947 return copy_memory_v(s, d, count, step);
948 }
949
950 bool is_backwards = step < 0;
951 int granularity = g_uabs(step);
952
953 const Register src = x30, dst = x31, cnt = x15, tmp3 = x16, tmp4 = x17, tmp5 = x14, tmp6 = x13;
954 const Register gct1 = x28, gct2 = x29, gct3 = t2;
955
956 Label same_aligned;
957 Label copy_big, copy32_loop, copy8_loop, copy_small, done;
958
959 // The size of copy32_loop body increases significantly with ZGC GC barriers.
960 // Need conditional far branches to reach a point beyond the loop in this case.
961 bool is_far = UseZGC;
962
963 __ beqz(count, done, is_far);
964 __ slli(cnt, count, exact_log2(granularity));
965 if (is_backwards) {
966 __ add(src, s, cnt);
967 __ add(dst, d, cnt);
968 } else {
969 __ mv(src, s);
970 __ mv(dst, d);
971 }
972
973 if (is_aligned) {
974 __ subi(t0, cnt, 32);
975 __ bgez(t0, copy32_loop);
976 __ subi(t0, cnt, 8);
977 __ bgez(t0, copy8_loop, is_far);
978 __ j(copy_small);
979 } else {
980 __ mv(t0, 16);
981 __ blt(cnt, t0, copy_small, is_far);
982
983 __ xorr(t0, src, dst);
984 __ andi(t0, t0, 0b111);
985 __ bnez(t0, copy_small, is_far);
986
987 __ bind(same_aligned);
988 __ andi(t0, src, 0b111);
989 __ beqz(t0, copy_big);
990 if (is_backwards) {
991 __ addi(src, src, step);
992 __ addi(dst, dst, step);
993 }
994 bs_asm->copy_load_at(_masm, decorators, type, granularity, tmp3, Address(src), gct1);
995 bs_asm->copy_store_at(_masm, decorators, type, granularity, Address(dst), tmp3, gct1, gct2, gct3);
996 if (!is_backwards) {
997 __ addi(src, src, step);
998 __ addi(dst, dst, step);
999 }
1000 __ subi(cnt, cnt, granularity);
1001 __ beqz(cnt, done, is_far);
1002 __ j(same_aligned);
1003
1004 __ bind(copy_big);
1005 __ mv(t0, 32);
1006 __ blt(cnt, t0, copy8_loop, is_far);
1007 }
1008
1009 __ bind(copy32_loop);
1010 if (is_backwards) {
1011 __ subi(src, src, wordSize * 4);
1012 __ subi(dst, dst, wordSize * 4);
1013 }
1014 // we first load 32 bytes, then write it, so the direction here doesn't matter
1015 bs_asm->copy_load_at(_masm, decorators, type, 8, tmp3, Address(src), gct1);
1016 bs_asm->copy_load_at(_masm, decorators, type, 8, tmp4, Address(src, 8), gct1);
1017 bs_asm->copy_load_at(_masm, decorators, type, 8, tmp5, Address(src, 16), gct1);
1018 bs_asm->copy_load_at(_masm, decorators, type, 8, tmp6, Address(src, 24), gct1);
1019
1020 bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst), tmp3, gct1, gct2, gct3);
1021 bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst, 8), tmp4, gct1, gct2, gct3);
1022 bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst, 16), tmp5, gct1, gct2, gct3);
1023 bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst, 24), tmp6, gct1, gct2, gct3);
1024
1025 if (!is_backwards) {
1026 __ addi(src, src, wordSize * 4);
1027 __ addi(dst, dst, wordSize * 4);
1028 }
1029 __ subi(t0, cnt, 32 + wordSize * 4);
1030 __ subi(cnt, cnt, wordSize * 4);
1031 __ bgez(t0, copy32_loop); // cnt >= 32, do next loop
1032
1033 __ beqz(cnt, done); // if that's all - done
1034
1035 __ subi(t0, cnt, 8); // if not - copy the reminder
1036 __ bltz(t0, copy_small); // cnt < 8, go to copy_small, else fall through to copy8_loop
1037
1038 __ bind(copy8_loop);
1039 if (is_backwards) {
1040 __ subi(src, src, wordSize);
1041 __ subi(dst, dst, wordSize);
1042 }
1043 bs_asm->copy_load_at(_masm, decorators, type, 8, tmp3, Address(src), gct1);
1044 bs_asm->copy_store_at(_masm, decorators, type, 8, Address(dst), tmp3, gct1, gct2, gct3);
1045
1046 if (!is_backwards) {
1047 __ addi(src, src, wordSize);
1048 __ addi(dst, dst, wordSize);
1049 }
1050 __ subi(t0, cnt, 8 + wordSize);
1051 __ subi(cnt, cnt, wordSize);
1052 __ bgez(t0, copy8_loop); // cnt >= 8, do next loop
1053
1054 __ beqz(cnt, done); // if that's all - done
1055
1056 __ bind(copy_small);
1057 if (is_backwards) {
1058 __ addi(src, src, step);
1059 __ addi(dst, dst, step);
1060 }
1061
1062 bs_asm->copy_load_at(_masm, decorators, type, granularity, tmp3, Address(src), gct1);
1063 bs_asm->copy_store_at(_masm, decorators, type, granularity, Address(dst), tmp3, gct1, gct2, gct3);
1064
1065 if (!is_backwards) {
1066 __ addi(src, src, step);
1067 __ addi(dst, dst, step);
1068 }
1069 __ subi(cnt, cnt, granularity);
1070 __ bgtz(cnt, copy_small);
1071
1072 __ bind(done);
1073 }
1074
1075 // Scan over array at a for count oops, verifying each one.
1076 // Preserves a and count, clobbers t0 and t1.
1077 void verify_oop_array(size_t size, Register a, Register count, Register temp) {
1078 Label loop, end;
1079 __ mv(t1, zr);
1080 __ slli(t0, count, exact_log2(size));
1081 __ bind(loop);
1082 __ bgeu(t1, t0, end);
1083
1084 __ add(temp, a, t1);
1085 if (size == (size_t)wordSize) {
1086 __ ld(temp, Address(temp, 0));
1087 __ verify_oop(temp);
1088 } else {
1089 __ lwu(temp, Address(temp, 0));
1090 __ decode_heap_oop(temp); // calls verify_oop
1091 }
1092 __ add(t1, t1, size);
1093 __ j(loop);
1094 __ bind(end);
1095 }
1096
1097 // Arguments:
1098 // stub_id - is used to name the stub and identify all details of
1099 // how to perform the copy.
1100 //
1101 // nopush_entry - is assigned to the stub's post push entry point
1102 // unless it is null
1103 //
1104 // Inputs:
1105 // c_rarg0 - source array address
1106 // c_rarg1 - destination array address
1107 // c_rarg2 - element count, treated as ssize_t, can be zero
1108 //
1109 // If 'from' and/or 'to' are aligned on 4-byte boundaries, we let
1110 // the hardware handle it. The two dwords within qwords that span
1111 // cache line boundaries will still be loaded and stored atomically.
1112 //
1113 // Side Effects: nopush_entry is set to the (post push) entry point
1114 // so it can be used by the corresponding conjoint
1115 // copy method
1116 //
1117 address generate_disjoint_copy(StubId stub_id, address* nopush_entry) {
1118 size_t size;
1119 bool aligned;
1120 bool is_oop;
1121 bool dest_uninitialized;
1122 switch (stub_id) {
1123 case StubId::stubgen_jbyte_disjoint_arraycopy_id:
1124 size = sizeof(jbyte);
1125 aligned = false;
1126 is_oop = false;
1127 dest_uninitialized = false;
1128 break;
1129 case StubId::stubgen_arrayof_jbyte_disjoint_arraycopy_id:
1130 size = sizeof(jbyte);
1131 aligned = true;
1132 is_oop = false;
1133 dest_uninitialized = false;
1134 break;
1135 case StubId::stubgen_jshort_disjoint_arraycopy_id:
1136 size = sizeof(jshort);
1137 aligned = false;
1138 is_oop = false;
1139 dest_uninitialized = false;
1140 break;
1141 case StubId::stubgen_arrayof_jshort_disjoint_arraycopy_id:
1142 size = sizeof(jshort);
1143 aligned = true;
1144 is_oop = false;
1145 dest_uninitialized = false;
1146 break;
1147 case StubId::stubgen_jint_disjoint_arraycopy_id:
1148 size = sizeof(jint);
1149 aligned = false;
1150 is_oop = false;
1151 dest_uninitialized = false;
1152 break;
1153 case StubId::stubgen_arrayof_jint_disjoint_arraycopy_id:
1154 size = sizeof(jint);
1155 aligned = true;
1156 is_oop = false;
1157 dest_uninitialized = false;
1158 break;
1159 case StubId::stubgen_jlong_disjoint_arraycopy_id:
1160 // since this is always aligned we can (should!) use the same
1161 // stub as for case arrayof_jlong_disjoint_arraycopy
1162 ShouldNotReachHere();
1163 break;
1164 case StubId::stubgen_arrayof_jlong_disjoint_arraycopy_id:
1165 size = sizeof(jlong);
1166 aligned = true;
1167 is_oop = false;
1168 dest_uninitialized = false;
1169 break;
1170 case StubId::stubgen_oop_disjoint_arraycopy_id:
1171 size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1172 aligned = !UseCompressedOops;
1173 is_oop = true;
1174 dest_uninitialized = false;
1175 break;
1176 case StubId::stubgen_arrayof_oop_disjoint_arraycopy_id:
1177 size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1178 aligned = !UseCompressedOops;
1179 is_oop = true;
1180 dest_uninitialized = false;
1181 break;
1182 case StubId::stubgen_oop_disjoint_arraycopy_uninit_id:
1183 size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1184 aligned = !UseCompressedOops;
1185 is_oop = true;
1186 dest_uninitialized = true;
1187 break;
1188 case StubId::stubgen_arrayof_oop_disjoint_arraycopy_uninit_id:
1189 size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1190 aligned = !UseCompressedOops;
1191 is_oop = true;
1192 dest_uninitialized = true;
1193 break;
1194 default:
1195 ShouldNotReachHere();
1196 break;
1197 }
1198
1199 const Register s = c_rarg0, d = c_rarg1, count = c_rarg2;
1200 RegSet saved_reg = RegSet::of(s, d, count);
1201 __ align(CodeEntryAlignment);
1202 StubCodeMark mark(this, stub_id);
1203 address start = __ pc();
1204 __ enter();
1205
1206 if (nopush_entry != nullptr) {
1207 *nopush_entry = __ pc();
1208 // caller can pass a 64-bit byte count here (from Unsafe.copyMemory)
1209 BLOCK_COMMENT("Entry:");
1210 }
1211
1212 DecoratorSet decorators = IN_HEAP | IS_ARRAY | ARRAYCOPY_DISJOINT;
1213 if (dest_uninitialized) {
1214 decorators |= IS_DEST_UNINITIALIZED;
1215 }
1216 if (aligned) {
1217 decorators |= ARRAYCOPY_ALIGNED;
1218 }
1219
1220 BarrierSetAssembler *bs = BarrierSet::barrier_set()->barrier_set_assembler();
1221 bs->arraycopy_prologue(_masm, decorators, is_oop, s, d, count, saved_reg);
1222
1223 if (is_oop) {
1224 // save regs before copy_memory
1225 __ push_reg(RegSet::of(d, count), sp);
1226 }
1227
1228 {
1229 // UnsafeMemoryAccess page error: continue after unsafe access
1230 bool add_entry = !is_oop && (!aligned || sizeof(jlong) == size);
1231 UnsafeMemoryAccessMark umam(this, add_entry, true);
1232 copy_memory(decorators, is_oop ? T_OBJECT : T_BYTE, aligned, s, d, count, size);
1233 }
1234
1235 if (is_oop) {
1236 __ pop_reg(RegSet::of(d, count), sp);
1237 if (VerifyOops) {
1238 verify_oop_array(size, d, count, t2);
1239 }
1240 }
1241
1242 bs->arraycopy_epilogue(_masm, decorators, is_oop, d, count, t0);
1243
1244 __ leave();
1245 __ mv(x10, zr); // return 0
1246 __ ret();
1247 return start;
1248 }
1249
1250 // Arguments:
1251 // stub_id - is used to name the stub and identify all details of
1252 // how to perform the copy.
1253 //
1254 // nooverlap_target - identifes the (post push) entry for the
1255 // corresponding disjoint copy routine which can be
1256 // jumped to if the ranges do not actually overlap
1257 //
1258 // nopush_entry - is assigned to the stub's post push entry point
1259 // unless it is null
1260 //
1261 // Inputs:
1262 // c_rarg0 - source array address
1263 // c_rarg1 - destination array address
1264 // c_rarg2 - element count, treated as ssize_t, can be zero
1265 //
1266 // If 'from' and/or 'to' are aligned on 4-byte boundaries, we let
1267 // the hardware handle it. The two dwords within qwords that span
1268 // cache line boundaries will still be loaded and stored atomically.
1269 //
1270 // Side Effects:
1271 // nopush_entry is set to the no-overlap entry point so it can be
1272 // used by some other conjoint copy method
1273 //
1274 address generate_conjoint_copy(StubId stub_id, address nooverlap_target, address *nopush_entry) {
1275 const Register s = c_rarg0, d = c_rarg1, count = c_rarg2;
1276 RegSet saved_regs = RegSet::of(s, d, count);
1277 int size;
1278 bool aligned;
1279 bool is_oop;
1280 bool dest_uninitialized;
1281 switch (stub_id) {
1282 case StubId::stubgen_jbyte_arraycopy_id:
1283 size = sizeof(jbyte);
1284 aligned = false;
1285 is_oop = false;
1286 dest_uninitialized = false;
1287 break;
1288 case StubId::stubgen_arrayof_jbyte_arraycopy_id:
1289 size = sizeof(jbyte);
1290 aligned = true;
1291 is_oop = false;
1292 dest_uninitialized = false;
1293 break;
1294 case StubId::stubgen_jshort_arraycopy_id:
1295 size = sizeof(jshort);
1296 aligned = false;
1297 is_oop = false;
1298 dest_uninitialized = false;
1299 break;
1300 case StubId::stubgen_arrayof_jshort_arraycopy_id:
1301 size = sizeof(jshort);
1302 aligned = true;
1303 is_oop = false;
1304 dest_uninitialized = false;
1305 break;
1306 case StubId::stubgen_jint_arraycopy_id:
1307 size = sizeof(jint);
1308 aligned = false;
1309 is_oop = false;
1310 dest_uninitialized = false;
1311 break;
1312 case StubId::stubgen_arrayof_jint_arraycopy_id:
1313 size = sizeof(jint);
1314 aligned = true;
1315 is_oop = false;
1316 dest_uninitialized = false;
1317 break;
1318 case StubId::stubgen_jlong_arraycopy_id:
1319 // since this is always aligned we can (should!) use the same
1320 // stub as for case arrayof_jlong_disjoint_arraycopy
1321 ShouldNotReachHere();
1322 break;
1323 case StubId::stubgen_arrayof_jlong_arraycopy_id:
1324 size = sizeof(jlong);
1325 aligned = true;
1326 is_oop = false;
1327 dest_uninitialized = false;
1328 break;
1329 case StubId::stubgen_oop_arraycopy_id:
1330 size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1331 aligned = !UseCompressedOops;
1332 is_oop = true;
1333 dest_uninitialized = false;
1334 break;
1335 case StubId::stubgen_arrayof_oop_arraycopy_id:
1336 size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1337 aligned = !UseCompressedOops;
1338 is_oop = true;
1339 dest_uninitialized = false;
1340 break;
1341 case StubId::stubgen_oop_arraycopy_uninit_id:
1342 size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1343 aligned = !UseCompressedOops;
1344 is_oop = true;
1345 dest_uninitialized = true;
1346 break;
1347 case StubId::stubgen_arrayof_oop_arraycopy_uninit_id:
1348 size = UseCompressedOops ? sizeof (jint) : sizeof (jlong);
1349 aligned = !UseCompressedOops;
1350 is_oop = true;
1351 dest_uninitialized = true;
1352 break;
1353 default:
1354 ShouldNotReachHere();
1355 }
1356
1357 StubCodeMark mark(this, stub_id);
1358 address start = __ pc();
1359 __ enter();
1360
1361 if (nopush_entry != nullptr) {
1362 *nopush_entry = __ pc();
1363 // caller can pass a 64-bit byte count here (from Unsafe.copyMemory)
1364 BLOCK_COMMENT("Entry:");
1365 }
1366
1367 // use fwd copy when (d-s) above_equal (count*size)
1368 __ sub(t0, d, s);
1369 __ slli(t1, count, exact_log2(size));
1370 Label L_continue;
1371 __ bltu(t0, t1, L_continue);
1372 __ j(RuntimeAddress(nooverlap_target));
1373 __ bind(L_continue);
1374
1375 DecoratorSet decorators = IN_HEAP | IS_ARRAY;
1376 if (dest_uninitialized) {
1377 decorators |= IS_DEST_UNINITIALIZED;
1378 }
1379 if (aligned) {
1380 decorators |= ARRAYCOPY_ALIGNED;
1381 }
1382
1383 BarrierSetAssembler *bs = BarrierSet::barrier_set()->barrier_set_assembler();
1384 bs->arraycopy_prologue(_masm, decorators, is_oop, s, d, count, saved_regs);
1385
1386 if (is_oop) {
1387 // save regs before copy_memory
1388 __ push_reg(RegSet::of(d, count), sp);
1389 }
1390
1391 {
1392 // UnsafeMemoryAccess page error: continue after unsafe access
1393 bool add_entry = !is_oop && (!aligned || sizeof(jlong) == size);
1394 UnsafeMemoryAccessMark umam(this, add_entry, true);
1395 copy_memory(decorators, is_oop ? T_OBJECT : T_BYTE, aligned, s, d, count, -size);
1396 }
1397
1398 if (is_oop) {
1399 __ pop_reg(RegSet::of(d, count), sp);
1400 if (VerifyOops) {
1401 verify_oop_array(size, d, count, t2);
1402 }
1403 }
1404 bs->arraycopy_epilogue(_masm, decorators, is_oop, d, count, t0);
1405 __ leave();
1406 __ mv(x10, zr); // return 0
1407 __ ret();
1408 return start;
1409 }
1410
1411 // Helper for generating a dynamic type check.
1412 // Smashes t0, t1.
1413 void generate_type_check(Register sub_klass,
1414 Register super_check_offset,
1415 Register super_klass,
1416 Register result,
1417 Register tmp1,
1418 Register tmp2,
1419 Label& L_success) {
1420 assert_different_registers(sub_klass, super_check_offset, super_klass);
1421
1422 BLOCK_COMMENT("type_check:");
1423
1424 Label L_miss;
1425
1426 __ check_klass_subtype_fast_path(sub_klass, super_klass, noreg, &L_success, &L_miss, nullptr, super_check_offset);
1427 __ check_klass_subtype_slow_path(sub_klass, super_klass, tmp1, tmp2, &L_success, nullptr);
1428
1429 // Fall through on failure!
1430 __ BIND(L_miss);
1431 }
1432
1433 //
1434 // Generate checkcasting array copy stub
1435 //
1436 // Input:
1437 // c_rarg0 - source array address
1438 // c_rarg1 - destination array address
1439 // c_rarg2 - element count, treated as ssize_t, can be zero
1440 // c_rarg3 - size_t ckoff (super_check_offset)
1441 // c_rarg4 - oop ckval (super_klass)
1442 //
1443 // Output:
1444 // x10 == 0 - success
1445 // x10 == -1^K - failure, where K is partial transfer count
1446 //
1447 address generate_checkcast_copy(StubId stub_id, address* nopush_entry) {
1448 bool dest_uninitialized;
1449 switch (stub_id) {
1450 case StubId::stubgen_checkcast_arraycopy_id:
1451 dest_uninitialized = false;
1452 break;
1453 case StubId::stubgen_checkcast_arraycopy_uninit_id:
1454 dest_uninitialized = true;
1455 break;
1456 default:
1457 ShouldNotReachHere();
1458 }
1459
1460 Label L_load_element, L_store_element, L_do_card_marks, L_done, L_done_pop;
1461
1462 // Input registers (after setup_arg_regs)
1463 const Register from = c_rarg0; // source array address
1464 const Register to = c_rarg1; // destination array address
1465 const Register count = c_rarg2; // elementscount
1466 const Register ckoff = c_rarg3; // super_check_offset
1467 const Register ckval = c_rarg4; // super_klass
1468
1469 RegSet wb_pre_saved_regs = RegSet::range(c_rarg0, c_rarg4);
1470
1471 // Registers used as temps (x7, x9, x18 are save-on-entry)
1472 const Register count_save = x19; // orig elementscount
1473 const Register start_to = x18; // destination array start address
1474 const Register copied_oop = x7; // actual oop copied
1475 const Register r9_klass = x9; // oop._klass
1476
1477 // Registers used as gc temps (x15, x16, x17 are save-on-call)
1478 const Register gct1 = x15, gct2 = x16, gct3 = x17;
1479
1480 //---------------------------------------------------------------
1481 // Assembler stub will be used for this call to arraycopy
1482 // if the two arrays are subtypes of Object[] but the
1483 // destination array type is not equal to or a supertype
1484 // of the source type. Each element must be separately
1485 // checked.
1486
1487 assert_different_registers(from, to, count, ckoff, ckval, start_to,
1488 copied_oop, r9_klass, count_save);
1489
1490 __ align(CodeEntryAlignment);
1491 StubCodeMark mark(this, stub_id);
1492 address start = __ pc();
1493
1494 __ enter(); // required for proper stackwalking of RuntimeStub frame
1495
1496 // Caller of this entry point must set up the argument registers.
1497 if (nopush_entry != nullptr) {
1498 *nopush_entry = __ pc();
1499 BLOCK_COMMENT("Entry:");
1500 }
1501
1502 // Empty array: Nothing to do
1503 __ beqz(count, L_done);
1504
1505 __ push_reg(RegSet::of(x7, x9, x18, x19), sp);
1506
1507 #ifdef ASSERT
1508 BLOCK_COMMENT("assert consistent ckoff/ckval");
1509 // The ckoff and ckval must be mutually consistent,
1510 // even though caller generates both.
1511 { Label L;
1512 int sco_offset = in_bytes(Klass::super_check_offset_offset());
1513 __ lwu(start_to, Address(ckval, sco_offset));
1514 __ beq(ckoff, start_to, L);
1515 __ stop("super_check_offset inconsistent");
1516 __ bind(L);
1517 }
1518 #endif //ASSERT
1519
1520 DecoratorSet decorators = IN_HEAP | IS_ARRAY | ARRAYCOPY_CHECKCAST | ARRAYCOPY_DISJOINT;
1521 if (dest_uninitialized) {
1522 decorators |= IS_DEST_UNINITIALIZED;
1523 }
1524
1525 bool is_oop = true;
1526 int element_size = UseCompressedOops ? 4 : 8;
1527
1528 BarrierSetAssembler *bs = BarrierSet::barrier_set()->barrier_set_assembler();
1529 bs->arraycopy_prologue(_masm, decorators, is_oop, from, to, count, wb_pre_saved_regs);
1530
1531 // save the original count
1532 __ mv(count_save, count);
1533
1534 // Copy from low to high addresses
1535 __ mv(start_to, to); // Save destination array start address
1536 __ j(L_load_element);
1537
1538 // ======== begin loop ========
1539 // (Loop is rotated; its entry is L_load_element.)
1540 // Loop control:
1541 // for count to 0 do
1542 // copied_oop = load_heap_oop(from++)
1543 // ... generate_type_check ...
1544 // store_heap_oop(to++, copied_oop)
1545 // end
1546
1547 __ align(OptoLoopAlignment);
1548
1549 __ BIND(L_store_element);
1550 bs->copy_store_at(_masm, decorators, T_OBJECT, element_size,
1551 Address(to, 0), copied_oop,
1552 gct1, gct2, gct3);
1553 __ addi(to, to, UseCompressedOops ? 4 : 8);
1554 __ subi(count, count, 1);
1555 __ beqz(count, L_do_card_marks);
1556
1557 // ======== loop entry is here ========
1558 __ BIND(L_load_element);
1559 bs->copy_load_at(_masm, decorators, T_OBJECT, element_size,
1560 copied_oop, Address(from, 0),
1561 gct1);
1562 __ addi(from, from, UseCompressedOops ? 4 : 8);
1563 __ beqz(copied_oop, L_store_element);
1564
1565 __ load_klass(r9_klass, copied_oop);// query the object klass
1566
1567 BLOCK_COMMENT("type_check:");
1568 generate_type_check(r9_klass, /*sub_klass*/
1569 ckoff, /*super_check_offset*/
1570 ckval, /*super_klass*/
1571 x10, /*result*/
1572 gct1, /*tmp1*/
1573 gct2, /*tmp2*/
1574 L_store_element);
1575
1576 // Fall through on failure!
1577
1578 // ======== end loop ========
1579
1580 // It was a real error; we must depend on the caller to finish the job.
1581 // Register count = remaining oops, count_orig = total oops.
1582 // Emit GC store barriers for the oops we have copied and report
1583 // their number to the caller.
1584
1585 __ sub(count, count_save, count); // K = partially copied oop count
1586 __ xori(count, count, -1); // report (-1^K) to caller
1587 __ beqz(count, L_done_pop);
1588
1589 __ BIND(L_do_card_marks);
1590 bs->arraycopy_epilogue(_masm, decorators, is_oop, start_to, count_save, t0);
1591
1592 __ bind(L_done_pop);
1593 __ pop_reg(RegSet::of(x7, x9, x18, x19), sp);
1594 inc_counter_np(SharedRuntime::_checkcast_array_copy_ctr);
1595
1596 __ bind(L_done);
1597 __ mv(x10, count);
1598 __ leave();
1599 __ ret();
1600
1601 return start;
1602 }
1603
1604 // Perform range checks on the proposed arraycopy.
1605 // Kills temp, but nothing else.
1606 // Also, clean the sign bits of src_pos and dst_pos.
1607 void arraycopy_range_checks(Register src, // source array oop (c_rarg0)
1608 Register src_pos, // source position (c_rarg1)
1609 Register dst, // destination array oo (c_rarg2)
1610 Register dst_pos, // destination position (c_rarg3)
1611 Register length,
1612 Register temp,
1613 Label& L_failed) {
1614 BLOCK_COMMENT("arraycopy_range_checks:");
1615
1616 assert_different_registers(t0, temp);
1617
1618 // if [src_pos + length > arrayOop(src)->length()] then FAIL
1619 __ lwu(t0, Address(src, arrayOopDesc::length_offset_in_bytes()));
1620 __ addw(temp, length, src_pos);
1621 __ bgtu(temp, t0, L_failed);
1622
1623 // if [dst_pos + length > arrayOop(dst)->length()] then FAIL
1624 __ lwu(t0, Address(dst, arrayOopDesc::length_offset_in_bytes()));
1625 __ addw(temp, length, dst_pos);
1626 __ bgtu(temp, t0, L_failed);
1627
1628 // Have to clean up high 32 bits of 'src_pos' and 'dst_pos'.
1629 __ zext(src_pos, src_pos, 32);
1630 __ zext(dst_pos, dst_pos, 32);
1631
1632 BLOCK_COMMENT("arraycopy_range_checks done");
1633 }
1634
1635 address generate_unsafecopy_common_error_exit() {
1636 address start = __ pc();
1637 __ mv(x10, 0);
1638 __ leave();
1639 __ ret();
1640 return start;
1641 }
1642
1643 //
1644 // Generate 'unsafe' set memory stub
1645 // Though just as safe as the other stubs, it takes an unscaled
1646 // size_t (# bytes) argument instead of an element count.
1647 //
1648 // Input:
1649 // c_rarg0 - destination array address
1650 // c_rarg1 - byte count (size_t)
1651 // c_rarg2 - byte value
1652 //
1653 address generate_unsafe_setmemory() {
1654 __ align(CodeEntryAlignment);
1655 StubId stub_id = StubId::stubgen_unsafe_setmemory_id;
1656 StubCodeMark mark(this, stub_id);
1657 address start = __ pc();
1658
1659 // bump this on entry, not on exit:
1660 // inc_counter_np(SharedRuntime::_unsafe_set_memory_ctr);
1661
1662 Label L_fill_elements;
1663
1664 const Register dest = c_rarg0;
1665 const Register count = c_rarg1;
1666 const Register value = c_rarg2;
1667 const Register cnt_words = x28; // temp register
1668 const Register tmp_reg = x29; // temp register
1669
1670 // Mark remaining code as such which performs Unsafe accesses.
1671 UnsafeMemoryAccessMark umam(this, true, false);
1672
1673 __ enter(); // required for proper stackwalking of RuntimeStub frame
1674
1675 // if count < 8, jump to L_fill_elements
1676 __ mv(tmp_reg, 8); // 8 bytes fill by element
1677 __ bltu(count, tmp_reg, L_fill_elements);
1678
1679 // Propagate byte to 64-bit width
1680 // 8 bit -> 16 bit
1681 __ zext(value, value, 8);
1682 __ slli(tmp_reg, value, 8);
1683 __ orr(value, value, tmp_reg);
1684 // 16 bit -> 32 bit
1685 __ slli(tmp_reg, value, 16);
1686 __ orr(value, value, tmp_reg);
1687 // 32 bit -> 64 bit
1688 __ slli(tmp_reg, value, 32);
1689 __ orr(value, value, tmp_reg);
1690
1691 // Align source address at 8 bytes address boundary.
1692 Label L_skip_align1, L_skip_align2, L_skip_align4;
1693 // One byte misalignment happens.
1694 __ test_bit(tmp_reg, dest, 0);
1695 __ beqz(tmp_reg, L_skip_align1);
1696 __ sb(value, Address(dest, 0));
1697 __ addi(dest, dest, 1);
1698 __ subi(count, count, 1);
1699
1700 __ bind(L_skip_align1);
1701 // Two bytes misalignment happens.
1702 __ test_bit(tmp_reg, dest, 1);
1703 __ beqz(tmp_reg, L_skip_align2);
1704 __ sh(value, Address(dest, 0));
1705 __ addi(dest, dest, 2);
1706 __ subi(count, count, 2);
1707
1708 __ bind(L_skip_align2);
1709 // Four bytes misalignment happens.
1710 __ test_bit(tmp_reg, dest, 2);
1711 __ beqz(tmp_reg, L_skip_align4);
1712 __ sw(value, Address(dest, 0));
1713 __ addi(dest, dest, 4);
1714 __ subi(count, count, 4);
1715 __ bind(L_skip_align4);
1716
1717 // Fill large chunks
1718 __ srli(cnt_words, count, 3); // number of words
1719 __ slli(tmp_reg, cnt_words, 3);
1720 __ sub(count, count, tmp_reg);
1721 {
1722 __ fill_words(dest, cnt_words, value);
1723 }
1724
1725 // Handle copies less than 8 bytes
1726 __ bind(L_fill_elements);
1727 Label L_fill_2, L_fill_1, L_exit;
1728 __ test_bit(tmp_reg, count, 2);
1729 __ beqz(tmp_reg, L_fill_2);
1730 __ sb(value, Address(dest, 0));
1731 __ sb(value, Address(dest, 1));
1732 __ sb(value, Address(dest, 2));
1733 __ sb(value, Address(dest, 3));
1734 __ addi(dest, dest, 4);
1735
1736 __ bind(L_fill_2);
1737 __ test_bit(tmp_reg, count, 1);
1738 __ beqz(tmp_reg, L_fill_1);
1739 __ sb(value, Address(dest, 0));
1740 __ sb(value, Address(dest, 1));
1741 __ addi(dest, dest, 2);
1742
1743 __ bind(L_fill_1);
1744 __ test_bit(tmp_reg, count, 0);
1745 __ beqz(tmp_reg, L_exit);
1746 __ sb(value, Address(dest, 0));
1747
1748 __ bind(L_exit);
1749 __ leave();
1750 __ ret();
1751
1752 return start;
1753 }
1754
1755 //
1756 // Generate 'unsafe' array copy stub
1757 // Though just as safe as the other stubs, it takes an unscaled
1758 // size_t argument instead of an element count.
1759 //
1760 // Input:
1761 // c_rarg0 - source array address
1762 // c_rarg1 - destination array address
1763 // c_rarg2 - byte count, treated as ssize_t, can be zero
1764 //
1765 // Examines the alignment of the operands and dispatches
1766 // to a long, int, short, or byte copy loop.
1767 //
1768 address generate_unsafe_copy(address byte_copy_entry,
1769 address short_copy_entry,
1770 address int_copy_entry,
1771 address long_copy_entry) {
1772 assert_cond(byte_copy_entry != nullptr && short_copy_entry != nullptr &&
1773 int_copy_entry != nullptr && long_copy_entry != nullptr);
1774 Label L_long_aligned, L_int_aligned, L_short_aligned;
1775 const Register s = c_rarg0, d = c_rarg1, count = c_rarg2;
1776
1777 __ align(CodeEntryAlignment);
1778 StubId stub_id = StubId::stubgen_unsafe_arraycopy_id;
1779 StubCodeMark mark(this, stub_id);
1780 address start = __ pc();
1781 __ enter(); // required for proper stackwalking of RuntimeStub frame
1782
1783 // bump this on entry, not on exit:
1784 inc_counter_np(SharedRuntime::_unsafe_array_copy_ctr);
1785
1786 __ orr(t0, s, d);
1787 __ orr(t0, t0, count);
1788
1789 __ andi(t0, t0, BytesPerLong - 1);
1790 __ beqz(t0, L_long_aligned);
1791 __ andi(t0, t0, BytesPerInt - 1);
1792 __ beqz(t0, L_int_aligned);
1793 __ test_bit(t0, t0, 0);
1794 __ beqz(t0, L_short_aligned);
1795 __ j(RuntimeAddress(byte_copy_entry));
1796
1797 __ BIND(L_short_aligned);
1798 __ srli(count, count, LogBytesPerShort); // size => short_count
1799 __ j(RuntimeAddress(short_copy_entry));
1800 __ BIND(L_int_aligned);
1801 __ srli(count, count, LogBytesPerInt); // size => int_count
1802 __ j(RuntimeAddress(int_copy_entry));
1803 __ BIND(L_long_aligned);
1804 __ srli(count, count, LogBytesPerLong); // size => long_count
1805 __ j(RuntimeAddress(long_copy_entry));
1806
1807 return start;
1808 }
1809
1810 //
1811 // Generate generic array copy stubs
1812 //
1813 // Input:
1814 // c_rarg0 - src oop
1815 // c_rarg1 - src_pos (32-bits)
1816 // c_rarg2 - dst oop
1817 // c_rarg3 - dst_pos (32-bits)
1818 // c_rarg4 - element count (32-bits)
1819 //
1820 // Output:
1821 // x10 == 0 - success
1822 // x10 == -1^K - failure, where K is partial transfer count
1823 //
1824 address generate_generic_copy(address byte_copy_entry, address short_copy_entry,
1825 address int_copy_entry, address oop_copy_entry,
1826 address long_copy_entry, address checkcast_copy_entry) {
1827 assert_cond(byte_copy_entry != nullptr && short_copy_entry != nullptr &&
1828 int_copy_entry != nullptr && oop_copy_entry != nullptr &&
1829 long_copy_entry != nullptr && checkcast_copy_entry != nullptr);
1830 Label L_failed, L_failed_0, L_objArray;
1831 Label L_copy_bytes, L_copy_shorts, L_copy_ints, L_copy_longs;
1832
1833 // Input registers
1834 const Register src = c_rarg0; // source array oop
1835 const Register src_pos = c_rarg1; // source position
1836 const Register dst = c_rarg2; // destination array oop
1837 const Register dst_pos = c_rarg3; // destination position
1838 const Register length = c_rarg4;
1839
1840 // Registers used as temps
1841 const Register dst_klass = c_rarg5;
1842
1843 __ align(CodeEntryAlignment);
1844
1845 StubId stub_id = StubId::stubgen_generic_arraycopy_id;
1846 StubCodeMark mark(this, stub_id);
1847
1848 address start = __ pc();
1849
1850 __ enter(); // required for proper stackwalking of RuntimeStub frame
1851
1852 // bump this on entry, not on exit:
1853 inc_counter_np(SharedRuntime::_generic_array_copy_ctr);
1854
1855 //-----------------------------------------------------------------------
1856 // Assembler stub will be used for this call to arraycopy
1857 // if the following conditions are met:
1858 //
1859 // (1) src and dst must not be null.
1860 // (2) src_pos must not be negative.
1861 // (3) dst_pos must not be negative.
1862 // (4) length must not be negative.
1863 // (5) src klass and dst klass should be the same and not null.
1864 // (6) src and dst should be arrays.
1865 // (7) src_pos + length must not exceed length of src.
1866 // (8) dst_pos + length must not exceed length of dst.
1867 //
1868
1869 // if src is null then return -1
1870 __ beqz(src, L_failed);
1871
1872 // if [src_pos < 0] then return -1
1873 __ sext(t0, src_pos, 32);
1874 __ bltz(t0, L_failed);
1875
1876 // if dst is null then return -1
1877 __ beqz(dst, L_failed);
1878
1879 // if [dst_pos < 0] then return -1
1880 __ sext(t0, dst_pos, 32);
1881 __ bltz(t0, L_failed);
1882
1883 // registers used as temp
1884 const Register scratch_length = x28; // elements count to copy
1885 const Register scratch_src_klass = x29; // array klass
1886 const Register lh = x30; // layout helper
1887
1888 // if [length < 0] then return -1
1889 __ sext(scratch_length, length, 32); // length (elements count, 32-bits value)
1890 __ bltz(scratch_length, L_failed);
1891
1892 __ load_narrow_klass(scratch_src_klass, src);
1893 #ifdef ASSERT
1894 {
1895 BLOCK_COMMENT("assert klasses not null {");
1896 Label L1, L2;
1897 __ bnez(scratch_src_klass, L2); // it is broken if klass is null
1898 __ bind(L1);
1899 __ stop("broken null klass");
1900 __ bind(L2);
1901 __ load_narrow_klass(t0, dst);
1902 __ beqz(t0, L1); // this would be broken also
1903 BLOCK_COMMENT("} assert klasses not null done");
1904 }
1905 #endif
1906 __ decode_klass_not_null(scratch_src_klass, t0);
1907
1908 // Load layout helper (32-bits)
1909 //
1910 // |array_tag| | header_size | element_type | |log2_element_size|
1911 // 32 30 24 16 8 2 0
1912 //
1913 // array_tag: typeArray = 0x3, objArray = 0x2, non-array = 0x0
1914 //
1915
1916 const int lh_offset = in_bytes(Klass::layout_helper_offset());
1917
1918 // Handle objArrays completely differently...
1919 const jint objArray_lh = Klass::array_layout_helper(T_OBJECT);
1920 __ lw(lh, Address(scratch_src_klass, lh_offset));
1921 __ mv(t0, objArray_lh);
1922 __ beq(lh, t0, L_objArray);
1923
1924 // if [src->klass() != dst->klass()] then return -1
1925 __ load_klass(t1, dst);
1926 __ bne(t1, scratch_src_klass, L_failed);
1927
1928 // if src->is_Array() isn't null then return -1
1929 // i.e. (lh >= 0)
1930 __ bgez(lh, L_failed);
1931
1932 // At this point, it is known to be a typeArray (array_tag 0x3).
1933 #ifdef ASSERT
1934 {
1935 BLOCK_COMMENT("assert primitive array {");
1936 Label L;
1937 __ mv(t1, (int32_t)(Klass::_lh_array_tag_type_value << Klass::_lh_array_tag_shift));
1938 __ bge(lh, t1, L);
1939 __ stop("must be a primitive array");
1940 __ bind(L);
1941 BLOCK_COMMENT("} assert primitive array done");
1942 }
1943 #endif
1944
1945 arraycopy_range_checks(src, src_pos, dst, dst_pos, scratch_length,
1946 t1, L_failed);
1947
1948 // TypeArrayKlass
1949 //
1950 // src_addr = (src + array_header_in_bytes()) + (src_pos << log2elemsize)
1951 // dst_addr = (dst + array_header_in_bytes()) + (dst_pos << log2elemsize)
1952 //
1953
1954 const Register t0_offset = t0; // array offset
1955 const Register x30_elsize = lh; // element size
1956
1957 // Get array_header_in_bytes()
1958 int lh_header_size_width = exact_log2(Klass::_lh_header_size_mask + 1);
1959 int lh_header_size_msb = Klass::_lh_header_size_shift + lh_header_size_width;
1960 __ slli(t0_offset, lh, XLEN - lh_header_size_msb); // left shift to remove 24 ~ 32;
1961 __ srli(t0_offset, t0_offset, XLEN - lh_header_size_width); // array_offset
1962
1963 __ add(src, src, t0_offset); // src array offset
1964 __ add(dst, dst, t0_offset); // dst array offset
1965 BLOCK_COMMENT("choose copy loop based on element size");
1966
1967 // next registers should be set before the jump to corresponding stub
1968 const Register from = c_rarg0; // source array address
1969 const Register to = c_rarg1; // destination array address
1970 const Register count = c_rarg2; // elements count
1971
1972 // 'from', 'to', 'count' registers should be set in such order
1973 // since they are the same as 'src', 'src_pos', 'dst'.
1974
1975 assert(Klass::_lh_log2_element_size_shift == 0, "fix this code");
1976
1977 // The possible values of elsize are 0-3, i.e. exact_log2(element
1978 // size in bytes). We do a simple bitwise binary search.
1979 __ BIND(L_copy_bytes);
1980 __ test_bit(t0, x30_elsize, 1);
1981 __ bnez(t0, L_copy_ints);
1982 __ test_bit(t0, x30_elsize, 0);
1983 __ bnez(t0, L_copy_shorts);
1984 __ add(from, src, src_pos); // src_addr
1985 __ add(to, dst, dst_pos); // dst_addr
1986 __ sext(count, scratch_length, 32); // length
1987 __ j(RuntimeAddress(byte_copy_entry));
1988
1989 __ BIND(L_copy_shorts);
1990 __ shadd(from, src_pos, src, t0, 1); // src_addr
1991 __ shadd(to, dst_pos, dst, t0, 1); // dst_addr
1992 __ sext(count, scratch_length, 32); // length
1993 __ j(RuntimeAddress(short_copy_entry));
1994
1995 __ BIND(L_copy_ints);
1996 __ test_bit(t0, x30_elsize, 0);
1997 __ bnez(t0, L_copy_longs);
1998 __ shadd(from, src_pos, src, t0, 2); // src_addr
1999 __ shadd(to, dst_pos, dst, t0, 2); // dst_addr
2000 __ sext(count, scratch_length, 32); // length
2001 __ j(RuntimeAddress(int_copy_entry));
2002
2003 __ BIND(L_copy_longs);
2004 #ifdef ASSERT
2005 {
2006 BLOCK_COMMENT("assert long copy {");
2007 Label L;
2008 __ andi(lh, lh, Klass::_lh_log2_element_size_mask); // lh -> x30_elsize
2009 __ sext(lh, lh, 32);
2010 __ mv(t0, LogBytesPerLong);
2011 __ beq(x30_elsize, t0, L);
2012 __ stop("must be long copy, but elsize is wrong");
2013 __ bind(L);
2014 BLOCK_COMMENT("} assert long copy done");
2015 }
2016 #endif
2017 __ shadd(from, src_pos, src, t0, 3); // src_addr
2018 __ shadd(to, dst_pos, dst, t0, 3); // dst_addr
2019 __ sext(count, scratch_length, 32); // length
2020 __ j(RuntimeAddress(long_copy_entry));
2021
2022 // ObjArrayKlass
2023 __ BIND(L_objArray);
2024 // live at this point: scratch_src_klass, scratch_length, src[_pos], dst[_pos]
2025
2026 Label L_plain_copy, L_checkcast_copy;
2027 // test array classes for subtyping
2028 __ load_klass(t2, dst);
2029 __ bne(scratch_src_klass, t2, L_checkcast_copy); // usual case is exact equality
2030
2031 // Identically typed arrays can be copied without element-wise checks.
2032 arraycopy_range_checks(src, src_pos, dst, dst_pos, scratch_length,
2033 t1, L_failed);
2034
2035 __ shadd(from, src_pos, src, t0, LogBytesPerHeapOop);
2036 __ addi(from, from, arrayOopDesc::base_offset_in_bytes(T_OBJECT));
2037 __ shadd(to, dst_pos, dst, t0, LogBytesPerHeapOop);
2038 __ addi(to, to, arrayOopDesc::base_offset_in_bytes(T_OBJECT));
2039 __ sext(count, scratch_length, 32); // length
2040 __ BIND(L_plain_copy);
2041 __ j(RuntimeAddress(oop_copy_entry));
2042
2043 __ BIND(L_checkcast_copy);
2044 // live at this point: scratch_src_klass, scratch_length, t2 (dst_klass)
2045 {
2046 // Before looking at dst.length, make sure dst is also an objArray.
2047 __ lwu(t0, Address(t2, lh_offset));
2048 __ mv(t1, objArray_lh);
2049 __ bne(t0, t1, L_failed);
2050
2051 // It is safe to examine both src.length and dst.length.
2052 arraycopy_range_checks(src, src_pos, dst, dst_pos, scratch_length,
2053 t2, L_failed);
2054
2055 __ load_klass(dst_klass, dst); // reload
2056
2057 // Marshal the base address arguments now, freeing registers.
2058 __ shadd(from, src_pos, src, t0, LogBytesPerHeapOop);
2059 __ addi(from, from, arrayOopDesc::base_offset_in_bytes(T_OBJECT));
2060 __ shadd(to, dst_pos, dst, t0, LogBytesPerHeapOop);
2061 __ addi(to, to, arrayOopDesc::base_offset_in_bytes(T_OBJECT));
2062 __ sext(count, length, 32); // length (reloaded)
2063 const Register sco_temp = c_rarg3; // this register is free now
2064 assert_different_registers(from, to, count, sco_temp,
2065 dst_klass, scratch_src_klass);
2066
2067 // Generate the type check.
2068 const int sco_offset = in_bytes(Klass::super_check_offset_offset());
2069 __ lwu(sco_temp, Address(dst_klass, sco_offset));
2070
2071 // Smashes t0, t1
2072 generate_type_check(scratch_src_klass, sco_temp, dst_klass, noreg, noreg, noreg, L_plain_copy);
2073
2074 // Fetch destination element klass from the ObjArrayKlass header.
2075 int ek_offset = in_bytes(ObjArrayKlass::element_klass_offset());
2076 __ ld(dst_klass, Address(dst_klass, ek_offset));
2077 __ lwu(sco_temp, Address(dst_klass, sco_offset));
2078
2079 // the checkcast_copy loop needs two extra arguments:
2080 assert(c_rarg3 == sco_temp, "#3 already in place");
2081 // Set up arguments for checkcast_copy_entry.
2082 __ mv(c_rarg4, dst_klass); // dst.klass.element_klass
2083 __ j(RuntimeAddress(checkcast_copy_entry));
2084 }
2085
2086 __ BIND(L_failed);
2087 __ mv(x10, -1);
2088 __ leave(); // required for proper stackwalking of RuntimeStub frame
2089 __ ret();
2090
2091 return start;
2092 }
2093
2094 //
2095 // Generate stub for array fill. If "aligned" is true, the
2096 // "to" address is assumed to be heapword aligned.
2097 //
2098 // Arguments for generated stub:
2099 // to: c_rarg0
2100 // value: c_rarg1
2101 // count: c_rarg2 treated as signed
2102 //
2103 address generate_fill(StubId stub_id) {
2104 BasicType t;
2105 bool aligned;
2106
2107 switch (stub_id) {
2108 case StubId::stubgen_jbyte_fill_id:
2109 t = T_BYTE;
2110 aligned = false;
2111 break;
2112 case StubId::stubgen_jshort_fill_id:
2113 t = T_SHORT;
2114 aligned = false;
2115 break;
2116 case StubId::stubgen_jint_fill_id:
2117 t = T_INT;
2118 aligned = false;
2119 break;
2120 case StubId::stubgen_arrayof_jbyte_fill_id:
2121 t = T_BYTE;
2122 aligned = true;
2123 break;
2124 case StubId::stubgen_arrayof_jshort_fill_id:
2125 t = T_SHORT;
2126 aligned = true;
2127 break;
2128 case StubId::stubgen_arrayof_jint_fill_id:
2129 t = T_INT;
2130 aligned = true;
2131 break;
2132 default:
2133 ShouldNotReachHere();
2134 };
2135
2136 __ align(CodeEntryAlignment);
2137 StubCodeMark mark(this, stub_id);
2138 address start = __ pc();
2139
2140 BLOCK_COMMENT("Entry:");
2141
2142 const Register to = c_rarg0; // source array address
2143 const Register value = c_rarg1; // value
2144 const Register count = c_rarg2; // elements count
2145
2146 const Register bz_base = x28; // base for block_zero routine
2147 const Register cnt_words = x29; // temp register
2148 const Register tmp_reg = t1;
2149
2150 __ enter();
2151
2152 Label L_fill_elements;
2153
2154 int shift = -1;
2155 switch (t) {
2156 case T_BYTE:
2157 shift = 0;
2158 // Short arrays (< 8 bytes) fill by element
2159 __ mv(tmp_reg, 8 >> shift);
2160 __ bltu(count, tmp_reg, L_fill_elements);
2161
2162 // Zero extend value
2163 // 8 bit -> 16 bit
2164 __ zext(value, value, 8);
2165 __ slli(tmp_reg, value, 8);
2166 __ orr(value, value, tmp_reg);
2167
2168 // 16 bit -> 32 bit
2169 __ slli(tmp_reg, value, 16);
2170 __ orr(value, value, tmp_reg);
2171 break;
2172 case T_SHORT:
2173 shift = 1;
2174 // Short arrays (< 8 bytes) fill by element
2175 __ mv(tmp_reg, 8 >> shift);
2176 __ bltu(count, tmp_reg, L_fill_elements);
2177
2178 // Zero extend value
2179 // 16 bit -> 32 bit
2180 __ zext(value, value, 16);
2181 __ slli(tmp_reg, value, 16);
2182 __ orr(value, value, tmp_reg);
2183 break;
2184 case T_INT:
2185 shift = 2;
2186 // Short arrays (< 8 bytes) fill by element
2187 __ mv(tmp_reg, 8 >> shift);
2188 __ bltu(count, tmp_reg, L_fill_elements);
2189 break;
2190 default: ShouldNotReachHere();
2191 }
2192
2193 // Align source address at 8 bytes address boundary.
2194 Label L_skip_align1, L_skip_align2, L_skip_align4;
2195 if (!aligned) {
2196 switch (t) {
2197 case T_BYTE:
2198 // One byte misalignment happens only for byte arrays.
2199 __ test_bit(tmp_reg, to, 0);
2200 __ beqz(tmp_reg, L_skip_align1);
2201 __ sb(value, Address(to, 0));
2202 __ addi(to, to, 1);
2203 __ subiw(count, count, 1);
2204 __ bind(L_skip_align1);
2205 // Fallthrough
2206 case T_SHORT:
2207 // Two bytes misalignment happens only for byte and short (char) arrays.
2208 __ test_bit(tmp_reg, to, 1);
2209 __ beqz(tmp_reg, L_skip_align2);
2210 __ sh(value, Address(to, 0));
2211 __ addi(to, to, 2);
2212 __ subiw(count, count, 2 >> shift);
2213 __ bind(L_skip_align2);
2214 // Fallthrough
2215 case T_INT:
2216 // Align to 8 bytes, we know we are 4 byte aligned to start.
2217 __ test_bit(tmp_reg, to, 2);
2218 __ beqz(tmp_reg, L_skip_align4);
2219 __ sw(value, Address(to, 0));
2220 __ addi(to, to, 4);
2221 __ subiw(count, count, 4 >> shift);
2222 __ bind(L_skip_align4);
2223 break;
2224 default: ShouldNotReachHere();
2225 }
2226 }
2227
2228 //
2229 // Fill large chunks
2230 //
2231 __ srliw(cnt_words, count, 3 - shift); // number of words
2232
2233 // 32 bit -> 64 bit
2234 __ zext(value, value, 32);
2235 __ slli(tmp_reg, value, 32);
2236 __ orr(value, value, tmp_reg);
2237
2238 __ slli(tmp_reg, cnt_words, 3 - shift);
2239 __ subw(count, count, tmp_reg);
2240 {
2241 __ fill_words(to, cnt_words, value);
2242 }
2243
2244 // Handle copies less than 8 bytes.
2245 // Address may not be heapword aligned.
2246 Label L_fill_1, L_fill_2, L_exit;
2247 __ bind(L_fill_elements);
2248 switch (t) {
2249 case T_BYTE:
2250 __ test_bit(tmp_reg, count, 2);
2251 __ beqz(tmp_reg, L_fill_2);
2252 __ sb(value, Address(to, 0));
2253 __ sb(value, Address(to, 1));
2254 __ sb(value, Address(to, 2));
2255 __ sb(value, Address(to, 3));
2256 __ addi(to, to, 4);
2257
2258 __ bind(L_fill_2);
2259 __ test_bit(tmp_reg, count, 1);
2260 __ beqz(tmp_reg, L_fill_1);
2261 __ sb(value, Address(to, 0));
2262 __ sb(value, Address(to, 1));
2263 __ addi(to, to, 2);
2264
2265 __ bind(L_fill_1);
2266 __ test_bit(tmp_reg, count, 0);
2267 __ beqz(tmp_reg, L_exit);
2268 __ sb(value, Address(to, 0));
2269 break;
2270 case T_SHORT:
2271 __ test_bit(tmp_reg, count, 1);
2272 __ beqz(tmp_reg, L_fill_2);
2273 __ sh(value, Address(to, 0));
2274 __ sh(value, Address(to, 2));
2275 __ addi(to, to, 4);
2276
2277 __ bind(L_fill_2);
2278 __ test_bit(tmp_reg, count, 0);
2279 __ beqz(tmp_reg, L_exit);
2280 __ sh(value, Address(to, 0));
2281 break;
2282 case T_INT:
2283 __ beqz(count, L_exit);
2284 __ sw(value, Address(to, 0));
2285 break;
2286 default: ShouldNotReachHere();
2287 }
2288 __ bind(L_exit);
2289 __ leave();
2290 __ ret();
2291
2292 return start;
2293 }
2294
2295 void generate_arraycopy_stubs() {
2296 // Some copy stubs publish a normal entry and then a 2nd 'fallback'
2297 // entry immediately following their stack push. This can be used
2298 // as a post-push branch target for compatible stubs when they
2299 // identify a special case that can be handled by the fallback
2300 // stub e.g a disjoint copy stub may be use as a special case
2301 // fallback for its compatible conjoint copy stub.
2302 //
2303 // A no push entry is always returned in the following local and
2304 // then published by assigning to the appropriate entry field in
2305 // class StubRoutines. The entry value is then passed to the
2306 // generator for the compatible stub. That means the entry must be
2307 // listed when saving to/restoring from the AOT cache, ensuring
2308 // that the inter-stub jumps are noted at AOT-cache save and
2309 // relocated at AOT cache load.
2310 address nopush_entry = nullptr;
2311
2312 // generate the common exit first so later stubs can rely on it if
2313 // they want an UnsafeMemoryAccess exit non-local to the stub
2314 StubRoutines::_unsafecopy_common_exit = generate_unsafecopy_common_error_exit();
2315 // register the stub as the default exit with class UnsafeMemoryAccess
2316 UnsafeMemoryAccess::set_common_exit_stub_pc(StubRoutines::_unsafecopy_common_exit);
2317
2318 // generate and publish riscv-specific bulk copy routines first
2319 // so we can call them from other copy stubs
2320 StubRoutines::riscv::_copy_byte_f = generate_copy_longs(StubId::stubgen_copy_byte_f_id, c_rarg0, c_rarg1, t1);
2321 StubRoutines::riscv::_copy_byte_b = generate_copy_longs(StubId::stubgen_copy_byte_b_id, c_rarg0, c_rarg1, t1);
2322
2323 StubRoutines::riscv::_zero_blocks = generate_zero_blocks();
2324
2325 //*** jbyte
2326 // Always need aligned and unaligned versions
2327 StubRoutines::_jbyte_disjoint_arraycopy = generate_disjoint_copy(StubId::stubgen_jbyte_disjoint_arraycopy_id, &nopush_entry);
2328 // disjoint nopush entry is needed by conjoint copy
2329 StubRoutines::_jbyte_disjoint_arraycopy_nopush = nopush_entry;
2330 StubRoutines::_jbyte_arraycopy = generate_conjoint_copy(StubId::stubgen_jbyte_arraycopy_id, StubRoutines::_jbyte_disjoint_arraycopy_nopush, &nopush_entry);
2331 // conjoint nopush entry is needed by generic/unsafe copy
2332 StubRoutines::_jbyte_arraycopy_nopush = nopush_entry;
2333 StubRoutines::_arrayof_jbyte_disjoint_arraycopy = generate_disjoint_copy(StubId::stubgen_arrayof_jbyte_disjoint_arraycopy_id, &nopush_entry);
2334 // disjoint arrayof nopush entry is needed by conjoint copy
2335 StubRoutines::_arrayof_jbyte_disjoint_arraycopy_nopush = nopush_entry;
2336 StubRoutines::_arrayof_jbyte_arraycopy = generate_conjoint_copy(StubId::stubgen_arrayof_jbyte_arraycopy_id, StubRoutines::_arrayof_jbyte_disjoint_arraycopy_nopush, nullptr);
2337
2338 //*** jshort
2339 // Always need aligned and unaligned versions
2340 StubRoutines::_jshort_disjoint_arraycopy = generate_disjoint_copy(StubId::stubgen_jshort_disjoint_arraycopy_id, &nopush_entry);
2341 // disjoint nopush entry is needed by conjoint copy
2342 StubRoutines::_jshort_disjoint_arraycopy_nopush = nopush_entry;
2343 StubRoutines::_jshort_arraycopy = generate_conjoint_copy(StubId::stubgen_jshort_arraycopy_id, StubRoutines::_jshort_disjoint_arraycopy_nopush, &nopush_entry);
2344 // conjoint nopush entry is used by generic/unsafe copy
2345 StubRoutines::_jshort_arraycopy_nopush = nopush_entry;
2346 StubRoutines::_arrayof_jshort_disjoint_arraycopy = generate_disjoint_copy(StubId::stubgen_arrayof_jshort_disjoint_arraycopy_id, &nopush_entry);
2347 // disjoint arrayof nopush entry is needed by conjoint copy
2348 StubRoutines::_arrayof_jshort_disjoint_arraycopy_nopush = nopush_entry;
2349 StubRoutines::_arrayof_jshort_arraycopy = generate_conjoint_copy(StubId::stubgen_arrayof_jshort_arraycopy_id, StubRoutines::_arrayof_jshort_disjoint_arraycopy_nopush, nullptr);
2350
2351 //*** jint
2352 // Aligned versions
2353 StubRoutines::_arrayof_jint_disjoint_arraycopy = generate_disjoint_copy(StubId::stubgen_arrayof_jint_disjoint_arraycopy_id, &nopush_entry);
2354 // disjoint arrayof nopush entry is needed by conjoint copy
2355 StubRoutines::_arrayof_jint_disjoint_arraycopy_nopush = nopush_entry;
2356 StubRoutines::_arrayof_jint_arraycopy = generate_conjoint_copy(StubId::stubgen_arrayof_jint_arraycopy_id, StubRoutines::_arrayof_jint_disjoint_arraycopy_nopush, nullptr);
2357 // In 64 bit we need both aligned and unaligned versions of jint arraycopy.
2358 // entry_jint_arraycopy always points to the unaligned version
2359 StubRoutines::_jint_disjoint_arraycopy = generate_disjoint_copy(StubId::stubgen_jint_disjoint_arraycopy_id, &nopush_entry);
2360 // disjoint nopush entry is needed by conjoint copy
2361 StubRoutines::_jint_disjoint_arraycopy_nopush = nopush_entry;
2362 StubRoutines::_jint_arraycopy = generate_conjoint_copy(StubId::stubgen_jint_arraycopy_id, StubRoutines::_jint_disjoint_arraycopy_nopush, &nopush_entry);
2363 // conjoint nopush entry is needed by generic/unsafe copy
2364 StubRoutines::_jint_arraycopy_nopush = nopush_entry;
2365
2366 //*** jlong
2367 // It is always aligned
2368 StubRoutines::_arrayof_jlong_disjoint_arraycopy = generate_disjoint_copy(StubId::stubgen_arrayof_jlong_disjoint_arraycopy_id, &nopush_entry);
2369 // disjoint arrayof nopush entry is needed by conjoint copy
2370 StubRoutines::_arrayof_jlong_disjoint_arraycopy_nopush = nopush_entry;
2371 StubRoutines::_arrayof_jlong_arraycopy = generate_conjoint_copy(StubId::stubgen_arrayof_jlong_arraycopy_id, StubRoutines::_arrayof_jlong_disjoint_arraycopy_nopush, &nopush_entry);
2372 // conjoint nopush entry is needed by generic/unsafe copy
2373 StubRoutines::_jlong_arraycopy_nopush = nopush_entry;
2374 // disjoint normal/nopush and conjoint normal entries are not
2375 // generated since the arrayof versions are the same
2376 StubRoutines::_jlong_disjoint_arraycopy = StubRoutines::_arrayof_jlong_disjoint_arraycopy;
2377 StubRoutines::_jlong_disjoint_arraycopy_nopush = StubRoutines::_arrayof_jlong_disjoint_arraycopy_nopush;
2378 StubRoutines::_jlong_arraycopy = StubRoutines::_arrayof_jlong_arraycopy;
2379
2380 //*** oops
2381 StubRoutines::_arrayof_oop_disjoint_arraycopy
2382 = generate_disjoint_copy(StubId::stubgen_arrayof_oop_disjoint_arraycopy_id, &nopush_entry);
2383 // disjoint arrayof nopush entry is needed by conjoint copy
2384 StubRoutines::_arrayof_oop_disjoint_arraycopy_nopush = nopush_entry;
2385 StubRoutines::_arrayof_oop_arraycopy
2386 = generate_conjoint_copy(StubId::stubgen_arrayof_oop_arraycopy_id, StubRoutines::_arrayof_oop_disjoint_arraycopy_nopush, &nopush_entry);
2387 // conjoint arrayof nopush entry is needed by generic/unsafe copy
2388 StubRoutines::_oop_arraycopy_nopush = nopush_entry;
2389 // Aligned versions without pre-barriers
2390 StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit
2391 = generate_disjoint_copy(StubId::stubgen_arrayof_oop_disjoint_arraycopy_uninit_id, &nopush_entry);
2392 // disjoint arrayof+uninit nopush entry is needed by conjoint copy
2393 StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit_nopush = nopush_entry;
2394
2395 // note that we don't need a returned nopush entry because the
2396 // generic/unsafe copy does not cater for uninit arrays.
2397 StubRoutines::_arrayof_oop_arraycopy_uninit
2398 = generate_conjoint_copy(StubId::stubgen_arrayof_oop_arraycopy_uninit_id, StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit_nopush, nullptr);
2399
2400 // for oop copies reuse arrayof entries for non-arrayof cases
2401 StubRoutines::_oop_disjoint_arraycopy = StubRoutines::_arrayof_oop_disjoint_arraycopy;
2402 StubRoutines::_oop_disjoint_arraycopy_nopush = StubRoutines::_arrayof_oop_disjoint_arraycopy_nopush;
2403 StubRoutines::_oop_arraycopy = StubRoutines::_arrayof_oop_arraycopy;
2404 StubRoutines::_oop_disjoint_arraycopy_uninit = StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit;
2405 StubRoutines::_oop_disjoint_arraycopy_uninit_nopush = StubRoutines::_arrayof_oop_disjoint_arraycopy_uninit_nopush;
2406 StubRoutines::_oop_arraycopy_uninit = StubRoutines::_arrayof_oop_arraycopy_uninit;
2407
2408 StubRoutines::_checkcast_arraycopy = generate_checkcast_copy(StubId::stubgen_checkcast_arraycopy_id, &nopush_entry);
2409 // checkcast nopush entry is needed by generic copy
2410 StubRoutines::_checkcast_arraycopy_nopush = nopush_entry;
2411 // note that we don't need a returned nopush entry because the
2412 // generic copy does not cater for uninit arrays.
2413 StubRoutines::_checkcast_arraycopy_uninit = generate_checkcast_copy(StubId::stubgen_checkcast_arraycopy_uninit_id, nullptr);
2414
2415
2416 // unsafe arraycopy may fallback on conjoint stubs
2417 StubRoutines::_unsafe_arraycopy = generate_unsafe_copy(StubRoutines::_jbyte_arraycopy_nopush,
2418 StubRoutines::_jshort_arraycopy_nopush,
2419 StubRoutines::_jint_arraycopy_nopush,
2420 StubRoutines::_jlong_arraycopy_nopush);
2421
2422 // generic arraycopy may fallback on conjoint stubs
2423 StubRoutines::_generic_arraycopy = generate_generic_copy(StubRoutines::_jbyte_arraycopy_nopush,
2424 StubRoutines::_jshort_arraycopy_nopush,
2425 StubRoutines::_jint_arraycopy_nopush,
2426 StubRoutines::_oop_arraycopy_nopush,
2427 StubRoutines::_jlong_arraycopy_nopush,
2428 StubRoutines::_checkcast_arraycopy_nopush);
2429
2430 StubRoutines::_jbyte_fill = generate_fill(StubId::stubgen_jbyte_fill_id);
2431 StubRoutines::_jshort_fill = generate_fill(StubId::stubgen_jshort_fill_id);
2432 StubRoutines::_jint_fill = generate_fill(StubId::stubgen_jint_fill_id);
2433 StubRoutines::_arrayof_jbyte_fill = generate_fill(StubId::stubgen_arrayof_jbyte_fill_id);
2434 StubRoutines::_arrayof_jshort_fill = generate_fill(StubId::stubgen_arrayof_jshort_fill_id);
2435 StubRoutines::_arrayof_jint_fill = generate_fill(StubId::stubgen_arrayof_jint_fill_id);
2436
2437 StubRoutines::_unsafe_setmemory = generate_unsafe_setmemory();
2438 }
2439
2440 void aes_load_keys(const Register &key, VectorRegister *working_vregs, int rounds) {
2441 const int step = 16;
2442 for (int i = 0; i < rounds; i++) {
2443 __ vle32_v(working_vregs[i], key);
2444 // The keys are stored in little-endian array, while we need
2445 // to operate in big-endian.
2446 // So performing an endian-swap here with vrev8.v instruction
2447 __ vrev8_v(working_vregs[i], working_vregs[i]);
2448 __ addi(key, key, step);
2449 }
2450 }
2451
2452 void aes_encrypt(const VectorRegister &res, VectorRegister *working_vregs, int rounds) {
2453 assert(rounds <= 15, "rounds should be less than or equal to working_vregs size");
2454
2455 __ vxor_vv(res, res, working_vregs[0]);
2456 for (int i = 1; i < rounds - 1; i++) {
2457 __ vaesem_vv(res, working_vregs[i]);
2458 }
2459 __ vaesef_vv(res, working_vregs[rounds - 1]);
2460 }
2461
2462 // Arguments:
2463 //
2464 // Inputs:
2465 // c_rarg0 - source byte array address
2466 // c_rarg1 - destination byte array address
2467 // c_rarg2 - sessionKe (key) in little endian int array
2468 //
2469 address generate_aescrypt_encryptBlock() {
2470 assert(UseAESIntrinsics, "need AES instructions (Zvkned extension) support");
2471
2472 __ align(CodeEntryAlignment);
2473 StubId stub_id = StubId::stubgen_aescrypt_encryptBlock_id;
2474 StubCodeMark mark(this, stub_id);
2475
2476 Label L_aes128, L_aes192;
2477
2478 const Register from = c_rarg0; // source array address
2479 const Register to = c_rarg1; // destination array address
2480 const Register key = c_rarg2; // key array address
2481 const Register keylen = c_rarg3;
2482
2483 VectorRegister working_vregs[] = {
2484 v4, v5, v6, v7, v8, v9, v10, v11,
2485 v12, v13, v14, v15, v16, v17, v18
2486 };
2487 const VectorRegister res = v19;
2488
2489 address start = __ pc();
2490 __ enter();
2491
2492 __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2493
2494 __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2495 __ vle32_v(res, from);
2496
2497 __ mv(t2, 52); // key length could be only {11, 13, 15} * 4 = {44, 52, 60}
2498 __ bltu(keylen, t2, L_aes128);
2499 __ beq(keylen, t2, L_aes192);
2500 // Else we fallthrough to the biggest case (256-bit key size)
2501
2502 // Note: the following function performs key += 15*16
2503 aes_load_keys(key, working_vregs, 15);
2504 aes_encrypt(res, working_vregs, 15);
2505 __ vse32_v(res, to);
2506 __ mv(c_rarg0, 0);
2507 __ leave();
2508 __ ret();
2509
2510 __ bind(L_aes192);
2511 // Note: the following function performs key += 13*16
2512 aes_load_keys(key, working_vregs, 13);
2513 aes_encrypt(res, working_vregs, 13);
2514 __ vse32_v(res, to);
2515 __ mv(c_rarg0, 0);
2516 __ leave();
2517 __ ret();
2518
2519 __ bind(L_aes128);
2520 // Note: the following function performs key += 11*16
2521 aes_load_keys(key, working_vregs, 11);
2522 aes_encrypt(res, working_vregs, 11);
2523 __ vse32_v(res, to);
2524 __ mv(c_rarg0, 0);
2525 __ leave();
2526 __ ret();
2527
2528 return start;
2529 }
2530
2531 void aes_decrypt(const VectorRegister &res, VectorRegister *working_vregs, int rounds) {
2532 assert(rounds <= 15, "rounds should be less than or equal to working_vregs size");
2533
2534 __ vxor_vv(res, res, working_vregs[rounds - 1]);
2535 for (int i = rounds - 2; i > 0; i--) {
2536 __ vaesdm_vv(res, working_vregs[i]);
2537 }
2538 __ vaesdf_vv(res, working_vregs[0]);
2539 }
2540
2541 // Arguments:
2542 //
2543 // Inputs:
2544 // c_rarg0 - source byte array address
2545 // c_rarg1 - destination byte array address
2546 // c_rarg2 - sessionKe (key) in little endian int array
2547 //
2548 address generate_aescrypt_decryptBlock() {
2549 assert(UseAESIntrinsics, "need AES instructions (Zvkned extension) support");
2550
2551 __ align(CodeEntryAlignment);
2552 StubId stub_id = StubId::stubgen_aescrypt_decryptBlock_id;
2553 StubCodeMark mark(this, stub_id);
2554
2555 Label L_aes128, L_aes192;
2556
2557 const Register from = c_rarg0; // source array address
2558 const Register to = c_rarg1; // destination array address
2559 const Register key = c_rarg2; // key array address
2560 const Register keylen = c_rarg3;
2561
2562 VectorRegister working_vregs[] = {
2563 v4, v5, v6, v7, v8, v9, v10, v11,
2564 v12, v13, v14, v15, v16, v17, v18
2565 };
2566 const VectorRegister res = v19;
2567
2568 address start = __ pc();
2569 __ enter(); // required for proper stackwalking of RuntimeStub frame
2570
2571 __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2572
2573 __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2574 __ vle32_v(res, from);
2575
2576 __ mv(t2, 52); // key length could be only {11, 13, 15} * 4 = {44, 52, 60}
2577 __ bltu(keylen, t2, L_aes128);
2578 __ beq(keylen, t2, L_aes192);
2579 // Else we fallthrough to the biggest case (256-bit key size)
2580
2581 // Note: the following function performs key += 15*16
2582 aes_load_keys(key, working_vregs, 15);
2583 aes_decrypt(res, working_vregs, 15);
2584 __ vse32_v(res, to);
2585 __ mv(c_rarg0, 0);
2586 __ leave();
2587 __ ret();
2588
2589 __ bind(L_aes192);
2590 // Note: the following function performs key += 13*16
2591 aes_load_keys(key, working_vregs, 13);
2592 aes_decrypt(res, working_vregs, 13);
2593 __ vse32_v(res, to);
2594 __ mv(c_rarg0, 0);
2595 __ leave();
2596 __ ret();
2597
2598 __ bind(L_aes128);
2599 // Note: the following function performs key += 11*16
2600 aes_load_keys(key, working_vregs, 11);
2601 aes_decrypt(res, working_vregs, 11);
2602 __ vse32_v(res, to);
2603 __ mv(c_rarg0, 0);
2604 __ leave();
2605 __ ret();
2606
2607 return start;
2608 }
2609
2610 void cipherBlockChaining_encryptAESCrypt(int round, Register from, Register to, Register key,
2611 Register rvec, Register input_len) {
2612 const Register len = x29;
2613
2614 VectorRegister working_vregs[] = {
2615 v1, v2, v3, v4, v5, v6, v7, v8,
2616 v9, v10, v11, v12, v13, v14, v15
2617 };
2618
2619 const unsigned int BLOCK_SIZE = 16;
2620
2621 __ mv(len, input_len);
2622 // load init rvec
2623 __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2624 __ vle32_v(v16, rvec);
2625
2626 aes_load_keys(key, working_vregs, round);
2627 Label L_enc_loop;
2628 __ bind(L_enc_loop);
2629 // Encrypt from source by block size
2630 __ vle32_v(v17, from);
2631 __ addi(from, from, BLOCK_SIZE);
2632 __ vxor_vv(v16, v16, v17);
2633 aes_encrypt(v16, working_vregs, round);
2634 __ vse32_v(v16, to);
2635 __ addi(to, to, BLOCK_SIZE);
2636 __ subi(len, len, BLOCK_SIZE);
2637 __ bnez(len, L_enc_loop);
2638
2639 // save current rvec and return
2640 __ vse32_v(v16, rvec);
2641 __ mv(x10, input_len);
2642 __ leave();
2643 __ ret();
2644 }
2645
2646 // Arguments:
2647 //
2648 // Inputs:
2649 // c_rarg0 - source byte array address
2650 // c_rarg1 - destination byte array address
2651 // c_rarg2 - K (key) in little endian int array
2652 // c_rarg3 - r vector byte array address
2653 // c_rarg4 - input length
2654 //
2655 // Output:
2656 // x10 - input length
2657 //
2658 address generate_cipherBlockChaining_encryptAESCrypt() {
2659 assert(UseAESIntrinsics, "need AES instructions (Zvkned extension) support");
2660 __ align(CodeEntryAlignment);
2661 StubId stub_id = StubId::stubgen_cipherBlockChaining_encryptAESCrypt_id;
2662 StubCodeMark mark(this, stub_id);
2663
2664 const Register from = c_rarg0;
2665 const Register to = c_rarg1;
2666 const Register key = c_rarg2;
2667 const Register rvec = c_rarg3;
2668 const Register input_len = c_rarg4;
2669
2670 const Register keylen = x28;
2671
2672 address start = __ pc();
2673 __ enter();
2674
2675 Label L_aes128, L_aes192;
2676 // Compute #rounds for AES based on the length of the key array
2677 __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2678 __ mv(t0, 52);
2679 __ bltu(keylen, t0, L_aes128);
2680 __ beq(keylen, t0, L_aes192);
2681 // Else we fallthrough to the biggest case (256-bit key size)
2682
2683 // Note: the following function performs key += 15*16
2684 cipherBlockChaining_encryptAESCrypt(15, from, to, key, rvec, input_len);
2685
2686 // Note: the following function performs key += 11*16
2687 __ bind(L_aes128);
2688 cipherBlockChaining_encryptAESCrypt(11, from, to, key, rvec, input_len);
2689
2690 // Note: the following function performs key += 13*16
2691 __ bind(L_aes192);
2692 cipherBlockChaining_encryptAESCrypt(13, from, to, key, rvec, input_len);
2693
2694 return start;
2695 }
2696
2697 void cipherBlockChaining_decryptAESCrypt(int round, Register from, Register to, Register key,
2698 Register rvec, Register input_len) {
2699 const Register len = x29;
2700
2701 VectorRegister working_vregs[] = {
2702 v1, v2, v3, v4, v5, v6, v7, v8,
2703 v9, v10, v11, v12, v13, v14, v15
2704 };
2705
2706 const unsigned int BLOCK_SIZE = 16;
2707
2708 __ mv(len, input_len);
2709 // load init rvec
2710 __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2711 __ vle32_v(v16, rvec);
2712
2713 aes_load_keys(key, working_vregs, round);
2714 Label L_dec_loop;
2715 // Decrypt from source by block size
2716 __ bind(L_dec_loop);
2717 __ vle32_v(v17, from);
2718 __ addi(from, from, BLOCK_SIZE);
2719 __ vmv_v_v(v18, v17);
2720 aes_decrypt(v17, working_vregs, round);
2721 __ vxor_vv(v17, v17, v16);
2722 __ vse32_v(v17, to);
2723 __ vmv_v_v(v16, v18);
2724 __ addi(to, to, BLOCK_SIZE);
2725 __ subi(len, len, BLOCK_SIZE);
2726 __ bnez(len, L_dec_loop);
2727
2728 // save current rvec and return
2729 __ vse32_v(v16, rvec);
2730 __ mv(x10, input_len);
2731 __ leave();
2732 __ ret();
2733 }
2734
2735 // Arguments:
2736 //
2737 // Inputs:
2738 // c_rarg0 - source byte array address
2739 // c_rarg1 - destination byte array address
2740 // c_rarg2 - K (key) in little endian int array
2741 // c_rarg3 - r vector byte array address
2742 // c_rarg4 - input length
2743 //
2744 // Output:
2745 // x10 - input length
2746 //
2747 address generate_cipherBlockChaining_decryptAESCrypt() {
2748 assert(UseAESIntrinsics, "need AES instructions (Zvkned extension) support");
2749 __ align(CodeEntryAlignment);
2750 StubId stub_id = StubId::stubgen_cipherBlockChaining_decryptAESCrypt_id;
2751 StubCodeMark mark(this, stub_id);
2752
2753 const Register from = c_rarg0;
2754 const Register to = c_rarg1;
2755 const Register key = c_rarg2;
2756 const Register rvec = c_rarg3;
2757 const Register input_len = c_rarg4;
2758
2759 const Register keylen = x28;
2760
2761 address start = __ pc();
2762 __ enter();
2763
2764 Label L_aes128, L_aes192, L_aes128_loop, L_aes192_loop, L_aes256_loop;
2765 // Compute #rounds for AES based on the length of the key array
2766 __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2767 __ mv(t0, 52);
2768 __ bltu(keylen, t0, L_aes128);
2769 __ beq(keylen, t0, L_aes192);
2770 // Else we fallthrough to the biggest case (256-bit key size)
2771
2772 // Note: the following function performs key += 15*16
2773 cipherBlockChaining_decryptAESCrypt(15, from, to, key, rvec, input_len);
2774
2775 // Note: the following function performs key += 11*16
2776 __ bind(L_aes128);
2777 cipherBlockChaining_decryptAESCrypt(11, from, to, key, rvec, input_len);
2778
2779 // Note: the following function performs key += 13*16
2780 __ bind(L_aes192);
2781 cipherBlockChaining_decryptAESCrypt(13, from, to, key, rvec, input_len);
2782
2783 return start;
2784 }
2785
2786 // Load big-endian 128-bit from memory.
2787 void be_load_counter_128(Register counter_hi, Register counter_lo, Register counter) {
2788 __ ld(counter_lo, Address(counter, 8)); // Load 128-bits from counter
2789 __ ld(counter_hi, Address(counter));
2790 __ rev8(counter_lo, counter_lo); // Convert big-endian to little-endian
2791 __ rev8(counter_hi, counter_hi);
2792 }
2793
2794 // Little-endian 128-bit + 64-bit -> 128-bit addition.
2795 void add_counter_128(Register counter_hi, Register counter_lo) {
2796 assert_different_registers(counter_hi, counter_lo, t0);
2797 __ addi(counter_lo, counter_lo, 1);
2798 __ seqz(t0, counter_lo); // Check for result overflow
2799 __ add(counter_hi, counter_hi, t0); // Add 1 if overflow otherwise 0
2800 }
2801
2802 // Store big-endian 128-bit to memory.
2803 void be_store_counter_128(Register counter_hi, Register counter_lo, Register counter) {
2804 assert_different_registers(counter_hi, counter_lo, t0, t1);
2805 __ rev8(t0, counter_lo); // Convert little-endian to big-endian
2806 __ rev8(t1, counter_hi);
2807 __ sd(t0, Address(counter, 8)); // Store 128-bits to counter
2808 __ sd(t1, Address(counter));
2809 }
2810
2811 void counterMode_AESCrypt(int round, Register in, Register out, Register key, Register counter,
2812 Register input_len, Register saved_encrypted_ctr, Register used_ptr) {
2813 // Algorithm:
2814 //
2815 // aes_load_keys();
2816 // load_counter_128(counter_hi, counter_lo, counter);
2817 //
2818 // L_next:
2819 // if (used >= BLOCK_SIZE) goto L_main_loop;
2820 //
2821 // L_encrypt_next:
2822 // *out = *in ^ saved_encrypted_ctr[used]);
2823 // out++; in++; used++; len--;
2824 // if (len == 0) goto L_exit;
2825 // goto L_next;
2826 //
2827 // L_main_loop:
2828 // if (len == 0) goto L_exit;
2829 // saved_encrypted_ctr = aes_encrypt(counter);
2830 //
2831 // add_counter_128(counter_hi, counter_lo);
2832 // be_store_counter_128(counter_hi, counter_lo, counter);
2833 // used = 0;
2834 //
2835 // if(len < BLOCK_SIZE) goto L_encrypt_next;
2836 //
2837 // v_in = load_16Byte(in);
2838 // v_out = load_16Byte(out);
2839 // v_saved_encrypted_ctr = load_16Byte(saved_encrypted_ctr);
2840 // v_out = v_in ^ v_saved_encrypted_ctr;
2841 // out += BLOCK_SIZE;
2842 // in += BLOCK_SIZE;
2843 // len -= BLOCK_SIZE;
2844 // used = BLOCK_SIZE;
2845 // goto L_main_loop;
2846 //
2847 //
2848 // L_exit:
2849 // store(used);
2850 // result = input_len
2851 // return result;
2852
2853 const Register used = x28;
2854 const Register len = x29;
2855 const Register counter_hi = x30;
2856 const Register counter_lo = x31;
2857 const Register block_size = t2;
2858
2859 const unsigned int BLOCK_SIZE = 16;
2860
2861 VectorRegister working_vregs[] = {
2862 v1, v2, v3, v4, v5, v6, v7, v8,
2863 v9, v10, v11, v12, v13, v14, v15
2864 };
2865
2866 __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
2867
2868 __ lwu(used, Address(used_ptr));
2869 __ mv(len, input_len);
2870 __ mv(block_size, BLOCK_SIZE);
2871
2872 // load keys to working_vregs according to round
2873 aes_load_keys(key, working_vregs, round);
2874
2875 // 128-bit big-endian load
2876 be_load_counter_128(counter_hi, counter_lo, counter);
2877
2878 Label L_next, L_encrypt_next, L_main_loop, L_exit;
2879 // Check the last saved_encrypted_ctr used value, we fall through
2880 // to L_encrypt_next when the used value lower than block_size
2881 __ bind(L_next);
2882 __ bgeu(used, block_size, L_main_loop);
2883
2884 // There is still data left fewer than block_size after L_main_loop
2885 // or last used, we encrypt them one by one.
2886 __ bind(L_encrypt_next);
2887 __ add(t0, saved_encrypted_ctr, used);
2888 __ lbu(t1, Address(t0));
2889 __ lbu(t0, Address(in));
2890 __ xorr(t1, t1, t0);
2891 __ sb(t1, Address(out));
2892 __ addi(in, in, 1);
2893 __ addi(out, out, 1);
2894 __ addi(used, used, 1);
2895 __ subi(len, len, 1);
2896 __ beqz(len, L_exit);
2897 __ j(L_next);
2898
2899 // We will calculate the next saved_encrypted_ctr and encrypt the blocks of data
2900 // one by one until there is less than a full block remaining if len not zero
2901 __ bind(L_main_loop);
2902 __ beqz(len, L_exit);
2903 __ vle32_v(v16, counter);
2904
2905 // encrypt counter according to round
2906 aes_encrypt(v16, working_vregs, round);
2907
2908 __ vse32_v(v16, saved_encrypted_ctr);
2909
2910 // 128-bit little-endian increment
2911 add_counter_128(counter_hi, counter_lo);
2912 // 128-bit big-endian store
2913 be_store_counter_128(counter_hi, counter_lo, counter);
2914
2915 __ mv(used, 0);
2916 // Check if we have a full block_size
2917 __ bltu(len, block_size, L_encrypt_next);
2918
2919 // We have one full block to encrypt at least
2920 __ vle32_v(v17, in);
2921 __ vxor_vv(v16, v16, v17);
2922 __ vse32_v(v16, out);
2923 __ add(out, out, block_size);
2924 __ add(in, in, block_size);
2925 __ sub(len, len, block_size);
2926 __ mv(used, block_size);
2927 __ j(L_main_loop);
2928
2929 __ bind(L_exit);
2930 __ sw(used, Address(used_ptr));
2931 __ mv(x10, input_len);
2932 __ leave();
2933 __ ret();
2934 };
2935
2936 // CTR AES crypt.
2937 // Arguments:
2938 //
2939 // Inputs:
2940 // c_rarg0 - source byte array address
2941 // c_rarg1 - destination byte array address
2942 // c_rarg2 - K (key) in little endian int array
2943 // c_rarg3 - counter vector byte array address
2944 // c_rarg4 - input length
2945 // c_rarg5 - saved encryptedCounter start
2946 // c_rarg6 - saved used length
2947 //
2948 // Output:
2949 // x10 - input length
2950 //
2951 address generate_counterMode_AESCrypt() {
2952 assert(UseAESCTRIntrinsics, "need AES instructions (Zvkned extension) and Zbb extension support");
2953
2954 __ align(CodeEntryAlignment);
2955 StubId stub_id = StubId::stubgen_counterMode_AESCrypt_id;
2956 StubCodeMark mark(this, stub_id);
2957
2958 const Register in = c_rarg0;
2959 const Register out = c_rarg1;
2960 const Register key = c_rarg2;
2961 const Register counter = c_rarg3;
2962 const Register input_len = c_rarg4;
2963 const Register saved_encrypted_ctr = c_rarg5;
2964 const Register used_len_ptr = c_rarg6;
2965
2966 const Register keylen = c_rarg7; // temporary register
2967
2968 const address start = __ pc();
2969 __ enter();
2970
2971 Label L_exit;
2972 __ beqz(input_len, L_exit);
2973
2974 Label L_aes128, L_aes192;
2975 // Compute #rounds for AES based on the length of the key array
2976 __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
2977 __ mv(t0, 52); // key length could be only {11, 13, 15} * 4 = {44, 52, 60}
2978 __ bltu(keylen, t0, L_aes128);
2979 __ beq(keylen, t0, L_aes192);
2980 // Else we fallthrough to the biggest case (256-bit key size)
2981
2982 // Note: the following function performs crypt with key += 15*16
2983 counterMode_AESCrypt(15, in, out, key, counter, input_len, saved_encrypted_ctr, used_len_ptr);
2984
2985 // Note: the following function performs crypt with key += 13*16
2986 __ bind(L_aes192);
2987 counterMode_AESCrypt(13, in, out, key, counter, input_len, saved_encrypted_ctr, used_len_ptr);
2988
2989 // Note: the following function performs crypt with key += 11*16
2990 __ bind(L_aes128);
2991 counterMode_AESCrypt(11, in, out, key, counter, input_len, saved_encrypted_ctr, used_len_ptr);
2992
2993 __ bind(L_exit);
2994 __ mv(x10, input_len);
2995 __ leave();
2996 __ ret();
2997
2998 return start;
2999 }
3000
3001 void ghash_loop(Register state, Register subkeyH, Register data, Register blocks,
3002 VectorRegister vtmp1, VectorRegister vtmp2, VectorRegister vtmp3) {
3003 VectorRegister partial_hash = vtmp1;
3004 VectorRegister hash_subkey = vtmp2;
3005 VectorRegister cipher_text = vtmp3;
3006
3007 const unsigned int BLOCK_SIZE = 16;
3008
3009 __ vsetivli(x0, 2, Assembler::e64, Assembler::m1);
3010 __ vle64_v(hash_subkey, subkeyH);
3011 __ vrev8_v(hash_subkey, hash_subkey);
3012 __ vle64_v(partial_hash, state);
3013 __ vrev8_v(partial_hash, partial_hash);
3014
3015 __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
3016 Label L_ghash_loop;
3017 __ bind(L_ghash_loop);
3018 __ vle32_v(cipher_text, data);
3019 __ addi(data, data, BLOCK_SIZE);
3020 __ vghsh_vv(partial_hash, hash_subkey, cipher_text);
3021 __ subi(blocks, blocks, 1);
3022 __ bnez(blocks, L_ghash_loop);
3023
3024 __ vsetivli(x0, 2, Assembler::e64, Assembler::m1);
3025 __ vrev8_v(partial_hash, partial_hash);
3026 __ vse64_v(partial_hash, state);
3027 }
3028
3029 /**
3030 * Arguments:
3031 *
3032 * Input:
3033 * c_rarg0 - current state address
3034 * c_rarg1 - H key address
3035 * c_rarg2 - data address
3036 * c_rarg3 - number of blocks
3037 *
3038 * Output:
3039 * Updated state at c_rarg0
3040 */
3041 address generate_ghash_processBlocks() {
3042 assert(UseGHASHIntrinsics, "need GHASH instructions (Zvkg extension) and Zvbb support");
3043
3044 __ align(CodeEntryAlignment);
3045 StubId stub_id = StubId::stubgen_ghash_processBlocks_id;
3046 StubCodeMark mark(this, stub_id);
3047
3048 address start = __ pc();
3049 __ enter();
3050
3051 Register state = c_rarg0;
3052 Register subkeyH = c_rarg1;
3053 Register data = c_rarg2;
3054 Register blocks = c_rarg3;
3055
3056 VectorRegister vtmp1 = v1;
3057 VectorRegister vtmp2 = v2;
3058 VectorRegister vtmp3 = v3;
3059
3060 ghash_loop(state, subkeyH, data, blocks, vtmp1, vtmp2, vtmp3);
3061
3062 __ leave();
3063 __ ret();
3064
3065 return start;
3066 }
3067
3068 void gcm_counterMode_AESCrypt_blocks(int round, Register in, Register out, Register key, Register counter,
3069 Register input_len, VectorRegister *working_vregs, Register blocks,
3070 VectorRegister vtmp1, VectorRegister vtmp2, VectorRegister vtmp3) {
3071 __ srli(blocks, input_len, 4);
3072
3073 const unsigned int BLOCK_SIZE = 16;
3074 const unsigned int MASK_VALUE = 0b1000; // we need {1, 0, 0, 0} mask value here
3075 __ vsetivli(x0, 1, Assembler::e8, Assembler::m1);
3076 __ vmv_v_i(v0, MASK_VALUE);
3077
3078 __ vsetivli(x0, 4, Assembler::e32, Assembler::m1);
3079 // load keys to working_vregs according to round
3080 aes_load_keys(key, working_vregs, round);
3081
3082 __ vle32_v(vtmp1, counter);
3083 Label L_aes_ctr_loop;
3084 __ bind(L_aes_ctr_loop);
3085 __ vmv_v_v(vtmp2, vtmp1);
3086 // encrypt counter according to round
3087 aes_encrypt(vtmp2, working_vregs, round);
3088 __ vle32_v(vtmp3, in);
3089 __ vxor_vv(vtmp2, vtmp2, vtmp3);
3090 __ vse32_v(vtmp2, out);
3091 __ addi(out, out, BLOCK_SIZE);
3092 __ addi(in, in, BLOCK_SIZE);
3093 __ sub(blocks, blocks, 1);
3094 __ vrev8_v(vtmp1, vtmp1, Assembler::VectorMask::v0_t);
3095 __ vadd_vi(vtmp1, vtmp1, 0x1, Assembler::VectorMask::v0_t);
3096 __ vrev8_v(vtmp1, vtmp1, Assembler::VectorMask::v0_t);
3097 __ bnez(blocks, L_aes_ctr_loop);
3098
3099 __ vse32_v(vtmp1, counter);
3100 }
3101
3102 void gcm_ghash_blocks(Register state, Register subkeyH, Register ct, Register input_len, Register blocks,
3103 VectorRegister vtmp1, VectorRegister vtmp2, VectorRegister vtmp3) {
3104 __ srli(blocks, input_len, 4);
3105
3106 ghash_loop(state, subkeyH, ct, blocks, vtmp1, vtmp2, vtmp3);
3107
3108 __ mv(x10, input_len);
3109 __ leave();
3110 __ ret();
3111 }
3112
3113
3114 // Vector AES Galois Counter Mode implementation. Parameters:
3115 //
3116 // in = c_rarg0
3117 // input_len = c_rarg1
3118 // ct = c_rarg2 - ciphertext that ghash will read (out for encrypt, in for decrypt)
3119 // out = c_rarg3
3120 // key = c_rarg4
3121 // state = c_rarg5 - GHASH.state
3122 // subkeyHtbl = c_rarg6 - powers of H
3123 // counter = c_rarg7 - 16 bytes of CTR
3124 // return - number of processed bytes
3125 address generate_galoisCounterMode_AESCrypt() {
3126 assert(UseGHASHIntrinsics, "need GHASH instructions (Zvkg extension) and Zvbb support");
3127 assert(UseAESCTRIntrinsics, "need AES instructions (Zvkned extension) and Zbb extension support");
3128
3129 __ align(CodeEntryAlignment);
3130 StubId stub_id = StubId::stubgen_galoisCounterMode_AESCrypt_id;
3131 StubCodeMark mark(this, stub_id);
3132
3133 const Register in = c_rarg0;
3134 const Register input_len = c_rarg1;
3135 const Register ct = c_rarg2;
3136 const Register out = c_rarg3;
3137 const Register key = c_rarg4;
3138 const Register state = c_rarg5;
3139 const Register subkeyHtbl = c_rarg6;
3140 const Register counter = c_rarg7;
3141
3142 const Register keylen = x28;
3143 const Register blocks = x29;
3144
3145 VectorRegister working_vregs[] = {
3146 v1, v2, v3, v4, v5, v6, v7, v8,
3147 v9, v10, v11, v12, v13, v14, v15
3148 };
3149
3150 VectorRegister vtmp1 = v16;
3151 VectorRegister vtmp2 = v17;
3152 VectorRegister vtmp3 = v18;
3153
3154 const address start = __ pc();
3155 __ enter();
3156
3157 Label L_exit;
3158 // Requires input_len (512) bytes to efficiently use the intrinsic
3159 __ andi(input_len, input_len, -512);
3160 __ beqz(input_len, L_exit);
3161
3162 Label L_aes128, L_aes192;
3163 // Compute #rounds for AES based on the length of the key array
3164 __ lwu(keylen, Address(key, arrayOopDesc::length_offset_in_bytes() - arrayOopDesc::base_offset_in_bytes(T_INT)));
3165 __ mv(t0, 52); // key length could be only {11, 13, 15} * 4 = {44, 52, 60}
3166 __ bltu(keylen, t0, L_aes128);
3167 __ beq(keylen, t0, L_aes192);
3168 // Else we fallthrough to the biggest case (256-bit key size)
3169
3170 // Note: the following function performs crypt with key += 15*16
3171 gcm_counterMode_AESCrypt_blocks(15, in, out, key, counter, input_len, working_vregs, blocks, vtmp1, vtmp2, vtmp3);
3172 gcm_ghash_blocks(state, subkeyHtbl, ct, input_len, blocks, vtmp1, vtmp2, vtmp3);
3173
3174 // Note: the following function performs crypt with key += 13*16
3175 __ bind(L_aes192);
3176 gcm_counterMode_AESCrypt_blocks(13, in, out, key, counter, input_len, working_vregs, blocks, vtmp1, vtmp2, vtmp3);
3177 gcm_ghash_blocks(state, subkeyHtbl, ct, input_len, blocks, vtmp1, vtmp2, vtmp3);
3178
3179 // Note: the following function performs crypt with key += 11*16
3180 __ bind(L_aes128);
3181 gcm_counterMode_AESCrypt_blocks(11, in, out, key, counter, input_len, working_vregs, blocks, vtmp1, vtmp2, vtmp3);
3182 gcm_ghash_blocks(state, subkeyHtbl, ct, input_len, blocks, vtmp1, vtmp2, vtmp3);
3183
3184 __ bind(L_exit);
3185 __ mv(x10, input_len);
3186 __ leave();
3187 __ ret();
3188
3189 return start;
3190 }
3191
3192 // code for comparing 8 characters of strings with Latin1 and Utf16 encoding
3193 void compare_string_8_x_LU(Register tmpL, Register tmpU,
3194 Register strL, Register strU, Label& DIFF) {
3195 const Register tmp = x30, tmpLval = x12;
3196
3197 int base_offset = arrayOopDesc::base_offset_in_bytes(T_BYTE);
3198 assert((base_offset % (UseCompactObjectHeaders ? 4 : 8)) == 0, "Must be");
3199
3200 #ifdef ASSERT
3201 if (AvoidUnalignedAccesses) {
3202 Label align_ok;
3203 __ andi(t0, strL, 0x7);
3204 __ beqz(t0, align_ok);
3205 __ stop("bad alignment");
3206 __ bind(align_ok);
3207 }
3208 #endif
3209 __ ld(tmpLval, Address(strL));
3210 __ addi(strL, strL, wordSize);
3211
3212 // compare first 4 characters
3213 __ load_long_misaligned(tmpU, Address(strU), tmp, (base_offset % 8) != 0 ? 4 : 8);
3214 __ addi(strU, strU, wordSize);
3215 __ inflate_lo32(tmpL, tmpLval);
3216 __ xorr(tmp, tmpU, tmpL);
3217 __ bnez(tmp, DIFF);
3218
3219 // compare second 4 characters
3220 __ load_long_misaligned(tmpU, Address(strU), tmp, (base_offset % 8) != 0 ? 4 : 8);
3221 __ addi(strU, strU, wordSize);
3222 __ inflate_hi32(tmpL, tmpLval);
3223 __ xorr(tmp, tmpU, tmpL);
3224 __ bnez(tmp, DIFF);
3225 }
3226
3227 // x10 = result
3228 // x11 = str1
3229 // x12 = cnt1
3230 // x13 = str2
3231 // x14 = cnt2
3232 // x28 = tmp1
3233 // x29 = tmp2
3234 // x30 = tmp3
3235 address generate_compare_long_string_different_encoding(StubId stub_id) {
3236 bool isLU;
3237 switch (stub_id) {
3238 case StubId::stubgen_compare_long_string_LU_id:
3239 isLU = true;
3240 break;
3241 case StubId::stubgen_compare_long_string_UL_id:
3242 isLU = false;
3243 break;
3244 default:
3245 ShouldNotReachHere();
3246 };
3247 __ align(CodeEntryAlignment);
3248 StubCodeMark mark(this, stub_id);
3249 address entry = __ pc();
3250 Label SMALL_LOOP, TAIL, LOAD_LAST, DONE, CALCULATE_DIFFERENCE;
3251 const Register result = x10, str1 = x11, str2 = x13, cnt2 = x14,
3252 tmp1 = x28, tmp2 = x29, tmp3 = x30, tmp4 = x12;
3253
3254 int base_offset = arrayOopDesc::base_offset_in_bytes(T_BYTE);
3255 assert((base_offset % (UseCompactObjectHeaders ? 4 : 8)) == 0, "Must be");
3256
3257 Register strU = isLU ? str2 : str1,
3258 strL = isLU ? str1 : str2,
3259 tmpU = isLU ? tmp2 : tmp1, // where to keep U for comparison
3260 tmpL = isLU ? tmp1 : tmp2; // where to keep L for comparison
3261
3262 if (AvoidUnalignedAccesses && (base_offset % 8) != 0) {
3263 // Load 4 bytes from strL to make sure main loop is 8-byte aligned
3264 // cnt2 is >= 68 here, no need to check it for >= 0
3265 __ lwu(tmpL, Address(strL));
3266 __ addi(strL, strL, wordSize / 2);
3267 __ load_long_misaligned(tmpU, Address(strU), tmp4, (base_offset % 8) != 0 ? 4 : 8);
3268 __ addi(strU, strU, wordSize);
3269 __ inflate_lo32(tmp3, tmpL);
3270 __ mv(tmpL, tmp3);
3271 __ xorr(tmp3, tmpU, tmpL);
3272 __ bnez(tmp3, CALCULATE_DIFFERENCE);
3273 __ subi(cnt2, cnt2, wordSize / 2);
3274 }
3275
3276 // we are now 8-bytes aligned on strL when AvoidUnalignedAccesses is true
3277 __ subi(cnt2, cnt2, wordSize * 2);
3278 __ bltz(cnt2, TAIL);
3279 __ bind(SMALL_LOOP); // smaller loop
3280 __ subi(cnt2, cnt2, wordSize * 2);
3281 compare_string_8_x_LU(tmpL, tmpU, strL, strU, CALCULATE_DIFFERENCE);
3282 compare_string_8_x_LU(tmpL, tmpU, strL, strU, CALCULATE_DIFFERENCE);
3283 __ bgez(cnt2, SMALL_LOOP);
3284 __ addi(t0, cnt2, wordSize * 2);
3285 __ beqz(t0, DONE);
3286 __ bind(TAIL); // 1..15 characters left
3287 // Aligned access. Load bytes in portions - 4, 2, 1.
3288
3289 __ addi(t0, cnt2, wordSize);
3290 __ addi(cnt2, cnt2, wordSize * 2); // amount of characters left to process
3291 __ bltz(t0, LOAD_LAST);
3292 // remaining characters are greater than or equals to 8, we can do one compare_string_8_x_LU
3293 compare_string_8_x_LU(tmpL, tmpU, strL, strU, CALCULATE_DIFFERENCE);
3294 __ subi(cnt2, cnt2, wordSize);
3295 __ beqz(cnt2, DONE); // no character left
3296 __ bind(LOAD_LAST); // cnt2 = 1..7 characters left
3297
3298 __ subi(cnt2, cnt2, wordSize); // cnt2 is now an offset in strL which points to last 8 bytes
3299 __ slli(t0, cnt2, 1); // t0 is now an offset in strU which points to last 16 bytes
3300 __ add(strL, strL, cnt2); // Address of last 8 bytes in Latin1 string
3301 __ add(strU, strU, t0); // Address of last 16 bytes in UTF-16 string
3302 __ load_int_misaligned(tmpL, Address(strL), t0, false);
3303 __ load_long_misaligned(tmpU, Address(strU), t0, 2);
3304 __ inflate_lo32(tmp3, tmpL);
3305 __ mv(tmpL, tmp3);
3306 __ xorr(tmp3, tmpU, tmpL);
3307 __ bnez(tmp3, CALCULATE_DIFFERENCE);
3308
3309 __ addi(strL, strL, wordSize / 2); // Address of last 4 bytes in Latin1 string
3310 __ addi(strU, strU, wordSize); // Address of last 8 bytes in UTF-16 string
3311 __ load_int_misaligned(tmpL, Address(strL), t0, false);
3312 __ load_long_misaligned(tmpU, Address(strU), t0, 2);
3313 __ inflate_lo32(tmp3, tmpL);
3314 __ mv(tmpL, tmp3);
3315 __ xorr(tmp3, tmpU, tmpL);
3316 __ bnez(tmp3, CALCULATE_DIFFERENCE);
3317 __ j(DONE); // no character left
3318
3319 // Find the first different characters in the longwords and
3320 // compute their difference.
3321 __ bind(CALCULATE_DIFFERENCE);
3322 // count bits of trailing zero chars
3323 __ ctzc_bits(tmp4, tmp3);
3324 __ srl(tmp1, tmp1, tmp4);
3325 __ srl(tmp2, tmp2, tmp4);
3326 __ zext(tmp1, tmp1, 16);
3327 __ zext(tmp2, tmp2, 16);
3328 __ sub(result, tmp1, tmp2);
3329 __ bind(DONE);
3330 __ ret();
3331 return entry;
3332 }
3333
3334 address generate_method_entry_barrier() {
3335 __ align(CodeEntryAlignment);
3336 StubId stub_id = StubId::stubgen_method_entry_barrier_id;
3337 StubCodeMark mark(this, stub_id);
3338
3339 Label deoptimize_label;
3340
3341 address start = __ pc();
3342
3343 BarrierSetAssembler* bs_asm = BarrierSet::barrier_set()->barrier_set_assembler();
3344
3345 if (bs_asm->nmethod_patching_type() == NMethodPatchingType::conc_instruction_and_data_patch) {
3346 BarrierSetNMethod* bs_nm = BarrierSet::barrier_set()->barrier_set_nmethod();
3347 Address thread_epoch_addr(xthread, in_bytes(bs_nm->thread_disarmed_guard_value_offset()) + 4);
3348 __ la(t1, ExternalAddress(bs_asm->patching_epoch_addr()));
3349 __ lwu(t1, t1);
3350 __ sw(t1, thread_epoch_addr);
3351 // There are two ways this can work:
3352 // - The writer did system icache shootdown after the instruction stream update.
3353 // Hence do nothing.
3354 // - The writer trust us to make sure our icache is in sync before entering.
3355 // Hence use cmodx fence (fence.i, may change).
3356 if (UseCtxFencei) {
3357 __ cmodx_fence();
3358 }
3359 __ membar(__ LoadLoad);
3360 }
3361
3362 __ set_last_Java_frame(sp, fp, ra);
3363
3364 __ enter();
3365 __ addi(t1, sp, wordSize);
3366
3367 __ subi(sp, sp, 4 * wordSize);
3368
3369 __ push_call_clobbered_registers();
3370
3371 __ mv(c_rarg0, t1);
3372 __ call_VM_leaf(CAST_FROM_FN_PTR(address, BarrierSetNMethod::nmethod_stub_entry_barrier), 1);
3373
3374 __ reset_last_Java_frame(true);
3375
3376 __ mv(t0, x10);
3377
3378 __ pop_call_clobbered_registers();
3379
3380 __ bnez(t0, deoptimize_label);
3381
3382 __ leave();
3383 __ ret();
3384
3385 __ BIND(deoptimize_label);
3386
3387 __ ld(t0, Address(sp, 0));
3388 __ ld(fp, Address(sp, wordSize));
3389 __ ld(ra, Address(sp, wordSize * 2));
3390 __ ld(t1, Address(sp, wordSize * 3));
3391
3392 __ mv(sp, t0);
3393 __ jr(t1);
3394
3395 return start;
3396 }
3397
3398 // x10 = result
3399 // x11 = str1
3400 // x12 = cnt1
3401 // x13 = str2
3402 // x14 = cnt2
3403 // x28 = tmp1
3404 // x29 = tmp2
3405 // x30 = tmp3
3406 // x31 = tmp4
3407 address generate_compare_long_string_same_encoding(StubId stub_id) {
3408 bool isLL;
3409 switch (stub_id) {
3410 case StubId::stubgen_compare_long_string_LL_id:
3411 isLL = true;
3412 break;
3413 case StubId::stubgen_compare_long_string_UU_id:
3414 isLL = false;
3415 break;
3416 default:
3417 ShouldNotReachHere();
3418 };
3419 __ align(CodeEntryAlignment);
3420 StubCodeMark mark(this, stub_id);
3421 address entry = __ pc();
3422 Label SMALL_LOOP, CHECK_LAST, DIFF2, TAIL,
3423 LENGTH_DIFF, DIFF, LAST_CHECK_AND_LENGTH_DIFF;
3424 const Register result = x10, str1 = x11, cnt1 = x12, str2 = x13, cnt2 = x14,
3425 tmp1 = x28, tmp2 = x29, tmp3 = x30, tmp4 = x7, tmp5 = x31;
3426 RegSet spilled_regs = RegSet::of(tmp4, tmp5);
3427
3428 // cnt1/cnt2 contains amount of characters to compare. cnt1 can be re-used
3429 // update cnt2 counter with already loaded 8 bytes
3430 __ subi(cnt2, cnt2, wordSize / (isLL ? 1 : 2));
3431 // update pointers, because of previous read
3432 __ addi(str1, str1, wordSize);
3433 __ addi(str2, str2, wordSize);
3434 // less than 16 bytes left?
3435 __ subi(cnt2, cnt2, isLL ? 16 : 8);
3436 __ push_reg(spilled_regs, sp);
3437 __ bltz(cnt2, TAIL);
3438 __ bind(SMALL_LOOP);
3439 // compare 16 bytes of strings with same encoding
3440 __ ld(tmp5, Address(str1));
3441 __ addi(str1, str1, 8);
3442 __ xorr(tmp4, tmp1, tmp2);
3443 __ ld(cnt1, Address(str2));
3444 __ addi(str2, str2, 8);
3445 __ bnez(tmp4, DIFF);
3446 __ ld(tmp1, Address(str1));
3447 __ addi(str1, str1, 8);
3448 __ xorr(tmp4, tmp5, cnt1);
3449 __ ld(tmp2, Address(str2));
3450 __ addi(str2, str2, 8);
3451 __ bnez(tmp4, DIFF2);
3452
3453 __ subi(cnt2, cnt2, isLL ? 16 : 8);
3454 __ bgez(cnt2, SMALL_LOOP);
3455 __ bind(TAIL);
3456 __ addi(cnt2, cnt2, isLL ? 16 : 8);
3457 __ beqz(cnt2, LAST_CHECK_AND_LENGTH_DIFF);
3458 __ subi(cnt2, cnt2, isLL ? 8 : 4);
3459 __ blez(cnt2, CHECK_LAST);
3460 __ xorr(tmp4, tmp1, tmp2);
3461 __ bnez(tmp4, DIFF);
3462 __ ld(tmp1, Address(str1));
3463 __ addi(str1, str1, 8);
3464 __ ld(tmp2, Address(str2));
3465 __ addi(str2, str2, 8);
3466 __ subi(cnt2, cnt2, isLL ? 8 : 4);
3467 __ bind(CHECK_LAST);
3468 if (!isLL) {
3469 __ add(cnt2, cnt2, cnt2); // now in bytes
3470 }
3471 __ xorr(tmp4, tmp1, tmp2);
3472 __ bnez(tmp4, DIFF);
3473 __ add(str1, str1, cnt2);
3474 __ load_long_misaligned(tmp5, Address(str1), tmp3, isLL ? 1 : 2);
3475 __ add(str2, str2, cnt2);
3476 __ load_long_misaligned(cnt1, Address(str2), tmp3, isLL ? 1 : 2);
3477 __ xorr(tmp4, tmp5, cnt1);
3478 __ beqz(tmp4, LENGTH_DIFF);
3479 // Find the first different characters in the longwords and
3480 // compute their difference.
3481 __ bind(DIFF2);
3482 // count bits of trailing zero chars
3483 __ ctzc_bits(tmp3, tmp4, isLL);
3484 __ srl(tmp5, tmp5, tmp3);
3485 __ srl(cnt1, cnt1, tmp3);
3486 if (isLL) {
3487 __ zext(tmp5, tmp5, 8);
3488 __ zext(cnt1, cnt1, 8);
3489 } else {
3490 __ zext(tmp5, tmp5, 16);
3491 __ zext(cnt1, cnt1, 16);
3492 }
3493 __ sub(result, tmp5, cnt1);
3494 __ j(LENGTH_DIFF);
3495 __ bind(DIFF);
3496 // count bits of trailing zero chars
3497 __ ctzc_bits(tmp3, tmp4, isLL);
3498 __ srl(tmp1, tmp1, tmp3);
3499 __ srl(tmp2, tmp2, tmp3);
3500 if (isLL) {
3501 __ zext(tmp1, tmp1, 8);
3502 __ zext(tmp2, tmp2, 8);
3503 } else {
3504 __ zext(tmp1, tmp1, 16);
3505 __ zext(tmp2, tmp2, 16);
3506 }
3507 __ sub(result, tmp1, tmp2);
3508 __ j(LENGTH_DIFF);
3509 __ bind(LAST_CHECK_AND_LENGTH_DIFF);
3510 __ xorr(tmp4, tmp1, tmp2);
3511 __ bnez(tmp4, DIFF);
3512 __ bind(LENGTH_DIFF);
3513 __ pop_reg(spilled_regs, sp);
3514 __ ret();
3515 return entry;
3516 }
3517
3518 void generate_compare_long_strings() {
3519 StubRoutines::riscv::_compare_long_string_LL = generate_compare_long_string_same_encoding(StubId::stubgen_compare_long_string_LL_id);
3520 StubRoutines::riscv::_compare_long_string_UU = generate_compare_long_string_same_encoding(StubId::stubgen_compare_long_string_UU_id);
3521 StubRoutines::riscv::_compare_long_string_LU = generate_compare_long_string_different_encoding(StubId::stubgen_compare_long_string_LU_id);
3522 StubRoutines::riscv::_compare_long_string_UL = generate_compare_long_string_different_encoding(StubId::stubgen_compare_long_string_UL_id);
3523 }
3524
3525 // x10 result
3526 // x11 src
3527 // x12 src count
3528 // x13 pattern
3529 // x14 pattern count
3530 address generate_string_indexof_linear(StubId stub_id)
3531 {
3532 bool needle_isL;
3533 bool haystack_isL;
3534 switch (stub_id) {
3535 case StubId::stubgen_string_indexof_linear_ll_id:
3536 needle_isL = true;
3537 haystack_isL = true;
3538 break;
3539 case StubId::stubgen_string_indexof_linear_ul_id:
3540 needle_isL = true;
3541 haystack_isL = false;
3542 break;
3543 case StubId::stubgen_string_indexof_linear_uu_id:
3544 needle_isL = false;
3545 haystack_isL = false;
3546 break;
3547 default:
3548 ShouldNotReachHere();
3549 };
3550
3551 __ align(CodeEntryAlignment);
3552 StubCodeMark mark(this, stub_id);
3553 address entry = __ pc();
3554
3555 int needle_chr_size = needle_isL ? 1 : 2;
3556 int haystack_chr_size = haystack_isL ? 1 : 2;
3557 int needle_chr_shift = needle_isL ? 0 : 1;
3558 int haystack_chr_shift = haystack_isL ? 0 : 1;
3559 bool isL = needle_isL && haystack_isL;
3560 // parameters
3561 Register result = x10, haystack = x11, haystack_len = x12, needle = x13, needle_len = x14;
3562 // temporary registers
3563 Register mask1 = x20, match_mask = x21, first = x22, trailing_zeros = x23, mask2 = x24, tmp = x25;
3564 // redefinitions
3565 Register ch1 = x28, ch2 = x29;
3566 RegSet spilled_regs = RegSet::range(x20, x25) + RegSet::range(x28, x29);
3567
3568 __ push_reg(spilled_regs, sp);
3569
3570 Label L_LOOP, L_LOOP_PROCEED, L_SMALL, L_HAS_ZERO,
3571 L_HAS_ZERO_LOOP, L_CMP_LOOP, L_CMP_LOOP_NOMATCH, L_SMALL_PROCEED,
3572 L_SMALL_HAS_ZERO_LOOP, L_SMALL_CMP_LOOP_NOMATCH, L_SMALL_CMP_LOOP,
3573 L_POST_LOOP, L_CMP_LOOP_LAST_CMP, L_HAS_ZERO_LOOP_NOMATCH,
3574 L_SMALL_CMP_LOOP_LAST_CMP, L_SMALL_CMP_LOOP_LAST_CMP2,
3575 L_CMP_LOOP_LAST_CMP2, DONE, NOMATCH;
3576
3577 __ ld(ch1, Address(needle));
3578 __ ld(ch2, Address(haystack));
3579 // src.length - pattern.length
3580 __ sub(haystack_len, haystack_len, needle_len);
3581
3582 // first is needle[0]
3583 __ zext(first, ch1, needle_isL ? 8 : 16);
3584
3585 uint64_t mask0101 = UCONST64(0x0101010101010101);
3586 uint64_t mask0001 = UCONST64(0x0001000100010001);
3587 __ mv(mask1, haystack_isL ? mask0101 : mask0001);
3588 __ mul(first, first, mask1);
3589 uint64_t mask7f7f = UCONST64(0x7f7f7f7f7f7f7f7f);
3590 uint64_t mask7fff = UCONST64(0x7fff7fff7fff7fff);
3591 __ mv(mask2, haystack_isL ? mask7f7f : mask7fff);
3592 if (needle_isL != haystack_isL) {
3593 __ mv(tmp, ch1);
3594 }
3595 __ subi(haystack_len, haystack_len, wordSize / haystack_chr_size - 1);
3596 __ blez(haystack_len, L_SMALL);
3597
3598 if (needle_isL != haystack_isL) {
3599 __ inflate_lo32(ch1, tmp, match_mask, trailing_zeros);
3600 }
3601 // xorr, sub, orr, notr, andr
3602 // compare and set match_mask[i] with 0x80/0x8000 (Latin1/UTF16) if ch2[i] == first[i]
3603 // eg:
3604 // first: aa aa aa aa aa aa aa aa
3605 // ch2: aa aa li nx jd ka aa aa
3606 // match_mask: 80 80 00 00 00 00 80 80
3607 __ compute_match_mask(ch2, first, match_mask, mask1, mask2);
3608
3609 // search first char of needle, if success, goto L_HAS_ZERO;
3610 __ bnez(match_mask, L_HAS_ZERO);
3611 __ subi(haystack_len, haystack_len, wordSize / haystack_chr_size);
3612 __ addi(result, result, wordSize / haystack_chr_size);
3613 __ addi(haystack, haystack, wordSize);
3614 __ bltz(haystack_len, L_POST_LOOP);
3615
3616 __ bind(L_LOOP);
3617 __ ld(ch2, Address(haystack));
3618 __ compute_match_mask(ch2, first, match_mask, mask1, mask2);
3619 __ bnez(match_mask, L_HAS_ZERO);
3620
3621 __ bind(L_LOOP_PROCEED);
3622 __ subi(haystack_len, haystack_len, wordSize / haystack_chr_size);
3623 __ addi(haystack, haystack, wordSize);
3624 __ addi(result, result, wordSize / haystack_chr_size);
3625 __ bgez(haystack_len, L_LOOP);
3626
3627 __ bind(L_POST_LOOP);
3628 __ mv(ch2, -wordSize / haystack_chr_size);
3629 __ ble(haystack_len, ch2, NOMATCH); // no extra characters to check
3630 __ ld(ch2, Address(haystack));
3631 __ slli(haystack_len, haystack_len, LogBitsPerByte + haystack_chr_shift);
3632 __ neg(haystack_len, haystack_len);
3633 __ xorr(ch2, first, ch2);
3634 __ sub(match_mask, ch2, mask1);
3635 __ orr(ch2, ch2, mask2);
3636 __ mv(trailing_zeros, -1); // all bits set
3637 __ j(L_SMALL_PROCEED);
3638
3639 __ align(OptoLoopAlignment);
3640 __ bind(L_SMALL);
3641 __ slli(haystack_len, haystack_len, LogBitsPerByte + haystack_chr_shift);
3642 __ neg(haystack_len, haystack_len);
3643 if (needle_isL != haystack_isL) {
3644 __ inflate_lo32(ch1, tmp, match_mask, trailing_zeros);
3645 }
3646 __ xorr(ch2, first, ch2);
3647 __ sub(match_mask, ch2, mask1);
3648 __ orr(ch2, ch2, mask2);
3649 __ mv(trailing_zeros, -1); // all bits set
3650
3651 __ bind(L_SMALL_PROCEED);
3652 __ srl(trailing_zeros, trailing_zeros, haystack_len); // mask. zeroes on useless bits.
3653 __ notr(ch2, ch2);
3654 __ andr(match_mask, match_mask, ch2);
3655 __ andr(match_mask, match_mask, trailing_zeros); // clear useless bits and check
3656 __ beqz(match_mask, NOMATCH);
3657
3658 __ bind(L_SMALL_HAS_ZERO_LOOP);
3659 // count bits of trailing zero chars
3660 __ ctzc_bits(trailing_zeros, match_mask, haystack_isL, ch2, tmp);
3661 __ addi(trailing_zeros, trailing_zeros, haystack_isL ? 7 : 15);
3662 __ mv(ch2, wordSize / haystack_chr_size);
3663 __ ble(needle_len, ch2, L_SMALL_CMP_LOOP_LAST_CMP2);
3664 __ compute_index(haystack, trailing_zeros, match_mask, result, ch2, tmp, haystack_isL);
3665 __ mv(trailing_zeros, wordSize / haystack_chr_size);
3666 __ bne(ch1, ch2, L_SMALL_CMP_LOOP_NOMATCH);
3667
3668 __ bind(L_SMALL_CMP_LOOP);
3669 __ shadd(first, trailing_zeros, needle, first, needle_chr_shift);
3670 __ shadd(ch2, trailing_zeros, haystack, ch2, haystack_chr_shift);
3671 needle_isL ? __ lbu(first, Address(first)) : __ lhu(first, Address(first));
3672 haystack_isL ? __ lbu(ch2, Address(ch2)) : __ lhu(ch2, Address(ch2));
3673 __ addi(trailing_zeros, trailing_zeros, 1);
3674 __ bge(trailing_zeros, needle_len, L_SMALL_CMP_LOOP_LAST_CMP);
3675 __ beq(first, ch2, L_SMALL_CMP_LOOP);
3676
3677 __ bind(L_SMALL_CMP_LOOP_NOMATCH);
3678 __ beqz(match_mask, NOMATCH);
3679 // count bits of trailing zero chars
3680 __ ctzc_bits(trailing_zeros, match_mask, haystack_isL, tmp, ch2);
3681 __ addi(trailing_zeros, trailing_zeros, haystack_isL ? 7 : 15);
3682 __ addi(result, result, 1);
3683 __ addi(haystack, haystack, haystack_chr_size);
3684 __ j(L_SMALL_HAS_ZERO_LOOP);
3685
3686 __ align(OptoLoopAlignment);
3687 __ bind(L_SMALL_CMP_LOOP_LAST_CMP);
3688 __ bne(first, ch2, L_SMALL_CMP_LOOP_NOMATCH);
3689 __ j(DONE);
3690
3691 __ align(OptoLoopAlignment);
3692 __ bind(L_SMALL_CMP_LOOP_LAST_CMP2);
3693 __ compute_index(haystack, trailing_zeros, match_mask, result, ch2, tmp, haystack_isL);
3694 __ bne(ch1, ch2, L_SMALL_CMP_LOOP_NOMATCH);
3695 __ j(DONE);
3696
3697 __ align(OptoLoopAlignment);
3698 __ bind(L_HAS_ZERO);
3699 // count bits of trailing zero chars
3700 __ ctzc_bits(trailing_zeros, match_mask, haystack_isL, tmp, ch2);
3701 __ addi(trailing_zeros, trailing_zeros, haystack_isL ? 7 : 15);
3702 __ slli(needle_len, needle_len, BitsPerByte * wordSize / 2);
3703 __ orr(haystack_len, haystack_len, needle_len); // restore needle_len(32bits)
3704 __ subi(result, result, 1); // array index from 0, so result -= 1
3705
3706 __ bind(L_HAS_ZERO_LOOP);
3707 __ mv(needle_len, wordSize / haystack_chr_size);
3708 __ srli(ch2, haystack_len, BitsPerByte * wordSize / 2);
3709 __ bge(needle_len, ch2, L_CMP_LOOP_LAST_CMP2);
3710 // load next 8 bytes from haystack, and increase result index
3711 __ compute_index(haystack, trailing_zeros, match_mask, result, ch2, tmp, haystack_isL);
3712 __ addi(result, result, 1);
3713 __ mv(trailing_zeros, wordSize / haystack_chr_size);
3714 __ bne(ch1, ch2, L_CMP_LOOP_NOMATCH);
3715
3716 // compare one char
3717 __ bind(L_CMP_LOOP);
3718 __ shadd(needle_len, trailing_zeros, needle, needle_len, needle_chr_shift);
3719 needle_isL ? __ lbu(needle_len, Address(needle_len)) : __ lhu(needle_len, Address(needle_len));
3720 __ shadd(ch2, trailing_zeros, haystack, ch2, haystack_chr_shift);
3721 haystack_isL ? __ lbu(ch2, Address(ch2)) : __ lhu(ch2, Address(ch2));
3722 __ addi(trailing_zeros, trailing_zeros, 1); // next char index
3723 __ srli(tmp, haystack_len, BitsPerByte * wordSize / 2);
3724 __ bge(trailing_zeros, tmp, L_CMP_LOOP_LAST_CMP);
3725 __ beq(needle_len, ch2, L_CMP_LOOP);
3726
3727 __ bind(L_CMP_LOOP_NOMATCH);
3728 __ beqz(match_mask, L_HAS_ZERO_LOOP_NOMATCH);
3729 // count bits of trailing zero chars
3730 __ ctzc_bits(trailing_zeros, match_mask, haystack_isL, needle_len, ch2);
3731 __ addi(trailing_zeros, trailing_zeros, haystack_isL ? 7 : 15);
3732 __ addi(haystack, haystack, haystack_chr_size);
3733 __ j(L_HAS_ZERO_LOOP);
3734
3735 __ align(OptoLoopAlignment);
3736 __ bind(L_CMP_LOOP_LAST_CMP);
3737 __ bne(needle_len, ch2, L_CMP_LOOP_NOMATCH);
3738 __ j(DONE);
3739
3740 __ align(OptoLoopAlignment);
3741 __ bind(L_CMP_LOOP_LAST_CMP2);
3742 __ compute_index(haystack, trailing_zeros, match_mask, result, ch2, tmp, haystack_isL);
3743 __ addi(result, result, 1);
3744 __ bne(ch1, ch2, L_CMP_LOOP_NOMATCH);
3745 __ j(DONE);
3746
3747 __ align(OptoLoopAlignment);
3748 __ bind(L_HAS_ZERO_LOOP_NOMATCH);
3749 // 1) Restore "result" index. Index was wordSize/str2_chr_size * N until
3750 // L_HAS_ZERO block. Byte octet was analyzed in L_HAS_ZERO_LOOP,
3751 // so, result was increased at max by wordSize/str2_chr_size - 1, so,
3752 // respective high bit wasn't changed. L_LOOP_PROCEED will increase
3753 // result by analyzed characters value, so, we can just reset lower bits
3754 // in result here. Clear 2 lower bits for UU/UL and 3 bits for LL
3755 // 2) restore needle_len and haystack_len values from "compressed" haystack_len
3756 // 3) advance haystack value to represent next haystack octet. result & 7/3 is
3757 // index of last analyzed substring inside current octet. So, haystack in at
3758 // respective start address. We need to advance it to next octet
3759 __ andi(match_mask, result, wordSize / haystack_chr_size - 1);
3760 __ srli(needle_len, haystack_len, BitsPerByte * wordSize / 2);
3761 __ andi(result, result, haystack_isL ? -8 : -4);
3762 __ slli(tmp, match_mask, haystack_chr_shift);
3763 __ sub(haystack, haystack, tmp);
3764 __ sext(haystack_len, haystack_len, 32);
3765 __ j(L_LOOP_PROCEED);
3766
3767 __ align(OptoLoopAlignment);
3768 __ bind(NOMATCH);
3769 __ mv(result, -1);
3770
3771 __ bind(DONE);
3772 __ pop_reg(spilled_regs, sp);
3773 __ ret();
3774 return entry;
3775 }
3776
3777 void generate_string_indexof_stubs()
3778 {
3779 StubRoutines::riscv::_string_indexof_linear_ll = generate_string_indexof_linear(StubId::stubgen_string_indexof_linear_ll_id);
3780 StubRoutines::riscv::_string_indexof_linear_uu = generate_string_indexof_linear(StubId::stubgen_string_indexof_linear_uu_id);
3781 StubRoutines::riscv::_string_indexof_linear_ul = generate_string_indexof_linear(StubId::stubgen_string_indexof_linear_ul_id);
3782 }
3783
3784 #ifdef COMPILER2
3785 void generate_lookup_secondary_supers_table_stub() {
3786 StubId stub_id = StubId::stubgen_lookup_secondary_supers_table_id;
3787 StubCodeMark mark(this, stub_id);
3788
3789 const Register
3790 r_super_klass = x10,
3791 r_array_base = x11,
3792 r_array_length = x12,
3793 r_array_index = x13,
3794 r_sub_klass = x14,
3795 result = x15,
3796 r_bitmap = x16;
3797
3798 for (int slot = 0; slot < Klass::SECONDARY_SUPERS_TABLE_SIZE; slot++) {
3799 StubRoutines::_lookup_secondary_supers_table_stubs[slot] = __ pc();
3800 Label L_success;
3801 __ enter();
3802 __ lookup_secondary_supers_table_const(r_sub_klass, r_super_klass, result,
3803 r_array_base, r_array_length, r_array_index,
3804 r_bitmap, slot, /*stub_is_near*/true);
3805 __ leave();
3806 __ ret();
3807 }
3808 }
3809
3810 // Slow path implementation for UseSecondarySupersTable.
3811 address generate_lookup_secondary_supers_table_slow_path_stub() {
3812 StubId stub_id = StubId::stubgen_lookup_secondary_supers_table_slow_path_id;
3813 StubCodeMark mark(this, stub_id);
3814
3815 address start = __ pc();
3816 const Register
3817 r_super_klass = x10, // argument
3818 r_array_base = x11, // argument
3819 temp1 = x12, // tmp
3820 r_array_index = x13, // argument
3821 result = x15, // argument
3822 r_bitmap = x16; // argument
3823
3824
3825 __ lookup_secondary_supers_table_slow_path(r_super_klass, r_array_base, r_array_index, r_bitmap, result, temp1);
3826 __ ret();
3827
3828 return start;
3829 }
3830
3831 address generate_mulAdd()
3832 {
3833 __ align(CodeEntryAlignment);
3834 StubId stub_id = StubId::stubgen_mulAdd_id;
3835 StubCodeMark mark(this, stub_id);
3836
3837 address entry = __ pc();
3838
3839 const Register out = x10;
3840 const Register in = x11;
3841 const Register offset = x12;
3842 const Register len = x13;
3843 const Register k = x14;
3844 const Register tmp = x28;
3845
3846 BLOCK_COMMENT("Entry:");
3847 __ enter();
3848 __ mul_add(out, in, offset, len, k, tmp);
3849 __ leave();
3850 __ ret();
3851
3852 return entry;
3853 }
3854
3855 /**
3856 * Arguments:
3857 *
3858 * Input:
3859 * c_rarg0 - x address
3860 * c_rarg1 - x length
3861 * c_rarg2 - y address
3862 * c_rarg3 - y length
3863 * c_rarg4 - z address
3864 */
3865 address generate_multiplyToLen()
3866 {
3867 __ align(CodeEntryAlignment);
3868 StubId stub_id = StubId::stubgen_multiplyToLen_id;
3869 StubCodeMark mark(this, stub_id);
3870 address entry = __ pc();
3871
3872 const Register x = x10;
3873 const Register xlen = x11;
3874 const Register y = x12;
3875 const Register ylen = x13;
3876 const Register z = x14;
3877
3878 const Register tmp0 = x15;
3879 const Register tmp1 = x16;
3880 const Register tmp2 = x17;
3881 const Register tmp3 = x7;
3882 const Register tmp4 = x28;
3883 const Register tmp5 = x29;
3884 const Register tmp6 = x30;
3885 const Register tmp7 = x31;
3886
3887 BLOCK_COMMENT("Entry:");
3888 __ enter(); // required for proper stackwalking of RuntimeStub frame
3889 __ multiply_to_len(x, xlen, y, ylen, z, tmp0, tmp1, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7);
3890 __ leave(); // required for proper stackwalking of RuntimeStub frame
3891 __ ret();
3892
3893 return entry;
3894 }
3895
3896 address generate_squareToLen()
3897 {
3898 __ align(CodeEntryAlignment);
3899 StubId stub_id = StubId::stubgen_squareToLen_id;
3900 StubCodeMark mark(this, stub_id);
3901 address entry = __ pc();
3902
3903 const Register x = x10;
3904 const Register xlen = x11;
3905 const Register z = x12;
3906 const Register y = x14; // == x
3907 const Register ylen = x15; // == xlen
3908
3909 const Register tmp0 = x13; // zlen, unused
3910 const Register tmp1 = x16;
3911 const Register tmp2 = x17;
3912 const Register tmp3 = x7;
3913 const Register tmp4 = x28;
3914 const Register tmp5 = x29;
3915 const Register tmp6 = x30;
3916 const Register tmp7 = x31;
3917
3918 BLOCK_COMMENT("Entry:");
3919 __ enter();
3920 __ mv(y, x);
3921 __ mv(ylen, xlen);
3922 __ multiply_to_len(x, xlen, y, ylen, z, tmp0, tmp1, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7);
3923 __ leave();
3924 __ ret();
3925
3926 return entry;
3927 }
3928
3929 // Arguments:
3930 //
3931 // Input:
3932 // c_rarg0 - newArr address
3933 // c_rarg1 - oldArr address
3934 // c_rarg2 - newIdx
3935 // c_rarg3 - shiftCount
3936 // c_rarg4 - numIter
3937 //
3938 address generate_bigIntegerLeftShift() {
3939 __ align(CodeEntryAlignment);
3940 StubId stub_id = StubId::stubgen_bigIntegerLeftShiftWorker_id;
3941 StubCodeMark mark(this, stub_id);
3942 address entry = __ pc();
3943
3944 Label loop, exit;
3945
3946 Register newArr = c_rarg0;
3947 Register oldArr = c_rarg1;
3948 Register newIdx = c_rarg2;
3949 Register shiftCount = c_rarg3;
3950 Register numIter = c_rarg4;
3951
3952 Register shiftRevCount = c_rarg5;
3953 Register oldArrNext = t1;
3954
3955 __ beqz(numIter, exit);
3956 __ shadd(newArr, newIdx, newArr, t0, 2);
3957
3958 __ mv(shiftRevCount, 32);
3959 __ sub(shiftRevCount, shiftRevCount, shiftCount);
3960
3961 __ bind(loop);
3962 __ addi(oldArrNext, oldArr, 4);
3963 __ vsetvli(t0, numIter, Assembler::e32, Assembler::m4);
3964 __ vle32_v(v0, oldArr);
3965 __ vle32_v(v4, oldArrNext);
3966 __ vsll_vx(v0, v0, shiftCount);
3967 __ vsrl_vx(v4, v4, shiftRevCount);
3968 __ vor_vv(v0, v0, v4);
3969 __ vse32_v(v0, newArr);
3970 __ sub(numIter, numIter, t0);
3971 __ shadd(oldArr, t0, oldArr, t1, 2);
3972 __ shadd(newArr, t0, newArr, t1, 2);
3973 __ bnez(numIter, loop);
3974
3975 __ bind(exit);
3976 __ ret();
3977
3978 return entry;
3979 }
3980
3981 // Arguments:
3982 //
3983 // Input:
3984 // c_rarg0 - newArr address
3985 // c_rarg1 - oldArr address
3986 // c_rarg2 - newIdx
3987 // c_rarg3 - shiftCount
3988 // c_rarg4 - numIter
3989 //
3990 address generate_bigIntegerRightShift() {
3991 __ align(CodeEntryAlignment);
3992 StubId stub_id = StubId::stubgen_bigIntegerRightShiftWorker_id;
3993 StubCodeMark mark(this, stub_id);
3994 address entry = __ pc();
3995
3996 Label loop, exit;
3997
3998 Register newArr = c_rarg0;
3999 Register oldArr = c_rarg1;
4000 Register newIdx = c_rarg2;
4001 Register shiftCount = c_rarg3;
4002 Register numIter = c_rarg4;
4003 Register idx = numIter;
4004
4005 Register shiftRevCount = c_rarg5;
4006 Register oldArrNext = c_rarg6;
4007 Register newArrCur = t0;
4008 Register oldArrCur = t1;
4009
4010 __ beqz(idx, exit);
4011 __ shadd(newArr, newIdx, newArr, t0, 2);
4012
4013 __ mv(shiftRevCount, 32);
4014 __ sub(shiftRevCount, shiftRevCount, shiftCount);
4015
4016 __ bind(loop);
4017 __ vsetvli(t0, idx, Assembler::e32, Assembler::m4);
4018 __ sub(idx, idx, t0);
4019 __ shadd(oldArrNext, idx, oldArr, t1, 2);
4020 __ shadd(newArrCur, idx, newArr, t1, 2);
4021 __ addi(oldArrCur, oldArrNext, 4);
4022 __ vle32_v(v0, oldArrCur);
4023 __ vle32_v(v4, oldArrNext);
4024 __ vsrl_vx(v0, v0, shiftCount);
4025 __ vsll_vx(v4, v4, shiftRevCount);
4026 __ vor_vv(v0, v0, v4);
4027 __ vse32_v(v0, newArrCur);
4028 __ bnez(idx, loop);
4029
4030 __ bind(exit);
4031 __ ret();
4032
4033 return entry;
4034 }
4035 #endif
4036
4037 #ifdef COMPILER2
4038 class MontgomeryMultiplyGenerator : public MacroAssembler {
4039
4040 Register Pa_base, Pb_base, Pn_base, Pm_base, inv, Rlen, Ra, Rb, Rm, Rn,
4041 Pa, Pb, Pn, Pm, Rhi_ab, Rlo_ab, Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2, Ri, Rj;
4042
4043 RegSet _toSave;
4044 bool _squaring;
4045
4046 public:
4047 MontgomeryMultiplyGenerator (Assembler *as, bool squaring)
4048 : MacroAssembler(as->code()), _squaring(squaring) {
4049
4050 // Register allocation
4051
4052 RegSetIterator<Register> regs = RegSet::range(x10, x26).begin();
4053 Pa_base = *regs; // Argument registers
4054 if (squaring) {
4055 Pb_base = Pa_base;
4056 } else {
4057 Pb_base = *++regs;
4058 }
4059 Pn_base = *++regs;
4060 Rlen= *++regs;
4061 inv = *++regs;
4062 Pm_base = *++regs;
4063
4064 // Working registers:
4065 Ra = *++regs; // The current digit of a, b, n, and m.
4066 Rb = *++regs;
4067 Rm = *++regs;
4068 Rn = *++regs;
4069
4070 Pa = *++regs; // Pointers to the current/next digit of a, b, n, and m.
4071 Pb = *++regs;
4072 Pm = *++regs;
4073 Pn = *++regs;
4074
4075 tmp0 = *++regs; // Three registers which form a
4076 tmp1 = *++regs; // triple-precision accumuator.
4077 tmp2 = *++regs;
4078
4079 Ri = x6; // Inner and outer loop indexes.
4080 Rj = x7;
4081
4082 Rhi_ab = x28; // Product registers: low and high parts
4083 Rlo_ab = x29; // of a*b and m*n.
4084 Rhi_mn = x30;
4085 Rlo_mn = x31;
4086
4087 // x18 and up are callee-saved.
4088 _toSave = RegSet::range(x18, *regs) + Pm_base;
4089 }
4090
4091 private:
4092 void save_regs() {
4093 push_reg(_toSave, sp);
4094 }
4095
4096 void restore_regs() {
4097 pop_reg(_toSave, sp);
4098 }
4099
4100 template <typename T>
4101 void unroll_2(Register count, T block) {
4102 Label loop, end, odd;
4103 beqz(count, end);
4104 test_bit(t0, count, 0);
4105 bnez(t0, odd);
4106 align(16);
4107 bind(loop);
4108 (this->*block)();
4109 bind(odd);
4110 (this->*block)();
4111 subi(count, count, 2);
4112 bgtz(count, loop);
4113 bind(end);
4114 }
4115
4116 template <typename T>
4117 void unroll_2(Register count, T block, Register d, Register s, Register tmp) {
4118 Label loop, end, odd;
4119 beqz(count, end);
4120 test_bit(tmp, count, 0);
4121 bnez(tmp, odd);
4122 align(16);
4123 bind(loop);
4124 (this->*block)(d, s, tmp);
4125 bind(odd);
4126 (this->*block)(d, s, tmp);
4127 subi(count, count, 2);
4128 bgtz(count, loop);
4129 bind(end);
4130 }
4131
4132 void pre1(RegisterOrConstant i) {
4133 block_comment("pre1");
4134 // Pa = Pa_base;
4135 // Pb = Pb_base + i;
4136 // Pm = Pm_base;
4137 // Pn = Pn_base + i;
4138 // Ra = *Pa;
4139 // Rb = *Pb;
4140 // Rm = *Pm;
4141 // Rn = *Pn;
4142 if (i.is_register()) {
4143 slli(t0, i.as_register(), LogBytesPerWord);
4144 } else {
4145 mv(t0, i.as_constant());
4146 slli(t0, t0, LogBytesPerWord);
4147 }
4148
4149 mv(Pa, Pa_base);
4150 add(Pb, Pb_base, t0);
4151 mv(Pm, Pm_base);
4152 add(Pn, Pn_base, t0);
4153
4154 ld(Ra, Address(Pa));
4155 ld(Rb, Address(Pb));
4156 ld(Rm, Address(Pm));
4157 ld(Rn, Address(Pn));
4158
4159 // Zero the m*n result.
4160 mv(Rhi_mn, zr);
4161 mv(Rlo_mn, zr);
4162 }
4163
4164 // The core multiply-accumulate step of a Montgomery
4165 // multiplication. The idea is to schedule operations as a
4166 // pipeline so that instructions with long latencies (loads and
4167 // multiplies) have time to complete before their results are
4168 // used. This most benefits in-order implementations of the
4169 // architecture but out-of-order ones also benefit.
4170 void step() {
4171 block_comment("step");
4172 // MACC(Ra, Rb, tmp0, tmp1, tmp2);
4173 // Ra = *++Pa;
4174 // Rb = *--Pb;
4175 mulhu(Rhi_ab, Ra, Rb);
4176 mul(Rlo_ab, Ra, Rb);
4177 addi(Pa, Pa, wordSize);
4178 ld(Ra, Address(Pa));
4179 subi(Pb, Pb, wordSize);
4180 ld(Rb, Address(Pb));
4181 acc(Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2); // The pending m*n from the
4182 // previous iteration.
4183 // MACC(Rm, Rn, tmp0, tmp1, tmp2);
4184 // Rm = *++Pm;
4185 // Rn = *--Pn;
4186 mulhu(Rhi_mn, Rm, Rn);
4187 mul(Rlo_mn, Rm, Rn);
4188 addi(Pm, Pm, wordSize);
4189 ld(Rm, Address(Pm));
4190 subi(Pn, Pn, wordSize);
4191 ld(Rn, Address(Pn));
4192 acc(Rhi_ab, Rlo_ab, tmp0, tmp1, tmp2);
4193 }
4194
4195 void post1() {
4196 block_comment("post1");
4197
4198 // MACC(Ra, Rb, tmp0, tmp1, tmp2);
4199 // Ra = *++Pa;
4200 // Rb = *--Pb;
4201 mulhu(Rhi_ab, Ra, Rb);
4202 mul(Rlo_ab, Ra, Rb);
4203 acc(Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2); // The pending m*n
4204 acc(Rhi_ab, Rlo_ab, tmp0, tmp1, tmp2);
4205
4206 // *Pm = Rm = tmp0 * inv;
4207 mul(Rm, tmp0, inv);
4208 sd(Rm, Address(Pm));
4209
4210 // MACC(Rm, Rn, tmp0, tmp1, tmp2);
4211 // tmp0 = tmp1; tmp1 = tmp2; tmp2 = 0;
4212 mulhu(Rhi_mn, Rm, Rn);
4213
4214 #ifndef PRODUCT
4215 // assert(m[i] * n[0] + tmp0 == 0, "broken Montgomery multiply");
4216 {
4217 mul(Rlo_mn, Rm, Rn);
4218 add(Rlo_mn, tmp0, Rlo_mn);
4219 Label ok;
4220 beqz(Rlo_mn, ok);
4221 stop("broken Montgomery multiply");
4222 bind(ok);
4223 }
4224 #endif
4225 // We have very carefully set things up so that
4226 // m[i]*n[0] + tmp0 == 0 (mod b), so we don't have to calculate
4227 // the lower half of Rm * Rn because we know the result already:
4228 // it must be -tmp0. tmp0 + (-tmp0) must generate a carry iff
4229 // tmp0 != 0. So, rather than do a mul and an cad we just set
4230 // the carry flag iff tmp0 is nonzero.
4231 //
4232 // mul(Rlo_mn, Rm, Rn);
4233 // cad(zr, tmp0, Rlo_mn);
4234 subi(t0, tmp0, 1);
4235 sltu(t0, t0, tmp0); // Set carry iff tmp0 is nonzero
4236 cadc(tmp0, tmp1, Rhi_mn, t0);
4237 adc(tmp1, tmp2, zr, t0);
4238 mv(tmp2, zr);
4239 }
4240
4241 void pre2(Register i, Register len) {
4242 block_comment("pre2");
4243 // Pa = Pa_base + i-len;
4244 // Pb = Pb_base + len;
4245 // Pm = Pm_base + i-len;
4246 // Pn = Pn_base + len;
4247
4248 sub(Rj, i, len);
4249 // Rj == i-len
4250
4251 // Ra as temp register
4252 slli(Ra, Rj, LogBytesPerWord);
4253 add(Pa, Pa_base, Ra);
4254 add(Pm, Pm_base, Ra);
4255 slli(Ra, len, LogBytesPerWord);
4256 add(Pb, Pb_base, Ra);
4257 add(Pn, Pn_base, Ra);
4258
4259 // Ra = *++Pa;
4260 // Rb = *--Pb;
4261 // Rm = *++Pm;
4262 // Rn = *--Pn;
4263 addi(Pa, Pa, wordSize);
4264 ld(Ra, Address(Pa));
4265 subi(Pb, Pb, wordSize);
4266 ld(Rb, Address(Pb));
4267 addi(Pm, Pm, wordSize);
4268 ld(Rm, Address(Pm));
4269 subi(Pn, Pn, wordSize);
4270 ld(Rn, Address(Pn));
4271
4272 mv(Rhi_mn, zr);
4273 mv(Rlo_mn, zr);
4274 }
4275
4276 void post2(Register i, Register len) {
4277 block_comment("post2");
4278 sub(Rj, i, len);
4279
4280 cad(tmp0, tmp0, Rlo_mn, t0); // The pending m*n, low part
4281
4282 // As soon as we know the least significant digit of our result,
4283 // store it.
4284 // Pm_base[i-len] = tmp0;
4285 // Rj as temp register
4286 slli(Rj, Rj, LogBytesPerWord);
4287 add(Rj, Pm_base, Rj);
4288 sd(tmp0, Address(Rj));
4289
4290 // tmp0 = tmp1; tmp1 = tmp2; tmp2 = 0;
4291 cadc(tmp0, tmp1, Rhi_mn, t0); // The pending m*n, high part
4292 adc(tmp1, tmp2, zr, t0);
4293 mv(tmp2, zr);
4294 }
4295
4296 // A carry in tmp0 after Montgomery multiplication means that we
4297 // should subtract multiples of n from our result in m. We'll
4298 // keep doing that until there is no carry.
4299 void normalize(Register len) {
4300 block_comment("normalize");
4301 // while (tmp0)
4302 // tmp0 = sub(Pm_base, Pn_base, tmp0, len);
4303 Label loop, post, again;
4304 Register cnt = tmp1, i = tmp2; // Re-use registers; we're done with them now
4305 beqz(tmp0, post); {
4306 bind(again); {
4307 mv(i, zr);
4308 mv(cnt, len);
4309 slli(Rn, i, LogBytesPerWord);
4310 add(Rm, Pm_base, Rn);
4311 ld(Rm, Address(Rm));
4312 add(Rn, Pn_base, Rn);
4313 ld(Rn, Address(Rn));
4314 mv(t0, 1); // set carry flag, i.e. no borrow
4315 align(16);
4316 bind(loop); {
4317 notr(Rn, Rn);
4318 add(Rm, Rm, t0);
4319 add(Rm, Rm, Rn);
4320 sltu(t0, Rm, Rn);
4321 slli(Rn, i, LogBytesPerWord); // Rn as temp register
4322 add(Rn, Pm_base, Rn);
4323 sd(Rm, Address(Rn));
4324 addi(i, i, 1);
4325 slli(Rn, i, LogBytesPerWord);
4326 add(Rm, Pm_base, Rn);
4327 ld(Rm, Address(Rm));
4328 add(Rn, Pn_base, Rn);
4329 ld(Rn, Address(Rn));
4330 subi(cnt, cnt, 1);
4331 } bnez(cnt, loop);
4332 subi(tmp0, tmp0, 1);
4333 add(tmp0, tmp0, t0);
4334 } bnez(tmp0, again);
4335 } bind(post);
4336 }
4337
4338 // Move memory at s to d, reversing words.
4339 // Increments d to end of copied memory
4340 // Destroys tmp1, tmp2
4341 // Preserves len
4342 // Leaves s pointing to the address which was in d at start
4343 void reverse(Register d, Register s, Register len, Register tmp1, Register tmp2) {
4344 assert(tmp1->encoding() < x28->encoding(), "register corruption");
4345 assert(tmp2->encoding() < x28->encoding(), "register corruption");
4346
4347 shadd(s, len, s, tmp1, LogBytesPerWord);
4348 mv(tmp1, len);
4349 unroll_2(tmp1, &MontgomeryMultiplyGenerator::reverse1, d, s, tmp2);
4350 slli(tmp1, len, LogBytesPerWord);
4351 sub(s, d, tmp1);
4352 }
4353 // [63...0] -> [31...0][63...32]
4354 void reverse1(Register d, Register s, Register tmp) {
4355 subi(s, s, wordSize);
4356 ld(tmp, Address(s));
4357 ror(tmp, tmp, 32, t0);
4358 sd(tmp, Address(d));
4359 addi(d, d, wordSize);
4360 }
4361
4362 void step_squaring() {
4363 // An extra ACC
4364 step();
4365 acc(Rhi_ab, Rlo_ab, tmp0, tmp1, tmp2);
4366 }
4367
4368 void last_squaring(Register i) {
4369 Label dont;
4370 // if ((i & 1) == 0) {
4371 test_bit(t0, i, 0);
4372 bnez(t0, dont); {
4373 // MACC(Ra, Rb, tmp0, tmp1, tmp2);
4374 // Ra = *++Pa;
4375 // Rb = *--Pb;
4376 mulhu(Rhi_ab, Ra, Rb);
4377 mul(Rlo_ab, Ra, Rb);
4378 acc(Rhi_ab, Rlo_ab, tmp0, tmp1, tmp2);
4379 } bind(dont);
4380 }
4381
4382 void extra_step_squaring() {
4383 acc(Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2); // The pending m*n
4384
4385 // MACC(Rm, Rn, tmp0, tmp1, tmp2);
4386 // Rm = *++Pm;
4387 // Rn = *--Pn;
4388 mulhu(Rhi_mn, Rm, Rn);
4389 mul(Rlo_mn, Rm, Rn);
4390 addi(Pm, Pm, wordSize);
4391 ld(Rm, Address(Pm));
4392 subi(Pn, Pn, wordSize);
4393 ld(Rn, Address(Pn));
4394 }
4395
4396 void post1_squaring() {
4397 acc(Rhi_mn, Rlo_mn, tmp0, tmp1, tmp2); // The pending m*n
4398
4399 // *Pm = Rm = tmp0 * inv;
4400 mul(Rm, tmp0, inv);
4401 sd(Rm, Address(Pm));
4402
4403 // MACC(Rm, Rn, tmp0, tmp1, tmp2);
4404 // tmp0 = tmp1; tmp1 = tmp2; tmp2 = 0;
4405 mulhu(Rhi_mn, Rm, Rn);
4406
4407 #ifndef PRODUCT
4408 // assert(m[i] * n[0] + tmp0 == 0, "broken Montgomery multiply");
4409 {
4410 mul(Rlo_mn, Rm, Rn);
4411 add(Rlo_mn, tmp0, Rlo_mn);
4412 Label ok;
4413 beqz(Rlo_mn, ok); {
4414 stop("broken Montgomery multiply");
4415 } bind(ok);
4416 }
4417 #endif
4418 // We have very carefully set things up so that
4419 // m[i]*n[0] + tmp0 == 0 (mod b), so we don't have to calculate
4420 // the lower half of Rm * Rn because we know the result already:
4421 // it must be -tmp0. tmp0 + (-tmp0) must generate a carry iff
4422 // tmp0 != 0. So, rather than do a mul and a cad we just set
4423 // the carry flag iff tmp0 is nonzero.
4424 //
4425 // mul(Rlo_mn, Rm, Rn);
4426 // cad(zr, tmp, Rlo_mn);
4427 subi(t0, tmp0, 1);
4428 sltu(t0, t0, tmp0); // Set carry iff tmp0 is nonzero
4429 cadc(tmp0, tmp1, Rhi_mn, t0);
4430 adc(tmp1, tmp2, zr, t0);
4431 mv(tmp2, zr);
4432 }
4433
4434 // use t0 as carry
4435 void acc(Register Rhi, Register Rlo,
4436 Register tmp0, Register tmp1, Register tmp2) {
4437 cad(tmp0, tmp0, Rlo, t0);
4438 cadc(tmp1, tmp1, Rhi, t0);
4439 adc(tmp2, tmp2, zr, t0);
4440 }
4441
4442 public:
4443 /**
4444 * Fast Montgomery multiplication. The derivation of the
4445 * algorithm is in A Cryptographic Library for the Motorola
4446 * DSP56000, Dusse and Kaliski, Proc. EUROCRYPT 90, pp. 230-237.
4447 *
4448 * Arguments:
4449 *
4450 * Inputs for multiplication:
4451 * c_rarg0 - int array elements a
4452 * c_rarg1 - int array elements b
4453 * c_rarg2 - int array elements n (the modulus)
4454 * c_rarg3 - int length
4455 * c_rarg4 - int inv
4456 * c_rarg5 - int array elements m (the result)
4457 *
4458 * Inputs for squaring:
4459 * c_rarg0 - int array elements a
4460 * c_rarg1 - int array elements n (the modulus)
4461 * c_rarg2 - int length
4462 * c_rarg3 - int inv
4463 * c_rarg4 - int array elements m (the result)
4464 *
4465 */
4466 address generate_multiply() {
4467 Label argh, nothing;
4468 bind(argh);
4469 stop("MontgomeryMultiply total_allocation must be <= 8192");
4470
4471 align(CodeEntryAlignment);
4472 address entry = pc();
4473
4474 beqz(Rlen, nothing);
4475
4476 enter();
4477
4478 // Make room.
4479 mv(Ra, 512);
4480 bgt(Rlen, Ra, argh);
4481 slli(Ra, Rlen, exact_log2(4 * sizeof(jint)));
4482 sub(Ra, sp, Ra);
4483 andi(sp, Ra, -2 * wordSize);
4484
4485 srliw(Rlen, Rlen, 1); // length in longwords = len/2
4486
4487 {
4488 // Copy input args, reversing as we go. We use Ra as a
4489 // temporary variable.
4490 reverse(Ra, Pa_base, Rlen, Ri, Rj);
4491 if (!_squaring)
4492 reverse(Ra, Pb_base, Rlen, Ri, Rj);
4493 reverse(Ra, Pn_base, Rlen, Ri, Rj);
4494 }
4495
4496 // Push all call-saved registers and also Pm_base which we'll need
4497 // at the end.
4498 save_regs();
4499
4500 #ifndef PRODUCT
4501 // assert(inv * n[0] == -1UL, "broken inverse in Montgomery multiply");
4502 {
4503 ld(Rn, Address(Pn_base));
4504 mul(Rlo_mn, Rn, inv);
4505 mv(t0, -1);
4506 Label ok;
4507 beq(Rlo_mn, t0, ok);
4508 stop("broken inverse in Montgomery multiply");
4509 bind(ok);
4510 }
4511 #endif
4512
4513 mv(Pm_base, Ra);
4514
4515 mv(tmp0, zr);
4516 mv(tmp1, zr);
4517 mv(tmp2, zr);
4518
4519 block_comment("for (int i = 0; i < len; i++) {");
4520 mv(Ri, zr); {
4521 Label loop, end;
4522 bge(Ri, Rlen, end);
4523
4524 bind(loop);
4525 pre1(Ri);
4526
4527 block_comment(" for (j = i; j; j--) {"); {
4528 mv(Rj, Ri);
4529 unroll_2(Rj, &MontgomeryMultiplyGenerator::step);
4530 } block_comment(" } // j");
4531
4532 post1();
4533 addiw(Ri, Ri, 1);
4534 blt(Ri, Rlen, loop);
4535 bind(end);
4536 block_comment("} // i");
4537 }
4538
4539 block_comment("for (int i = len; i < 2*len; i++) {");
4540 mv(Ri, Rlen); {
4541 Label loop, end;
4542 slli(t0, Rlen, 1);
4543 bge(Ri, t0, end);
4544
4545 bind(loop);
4546 pre2(Ri, Rlen);
4547
4548 block_comment(" for (j = len*2-i-1; j; j--) {"); {
4549 slliw(Rj, Rlen, 1);
4550 subw(Rj, Rj, Ri);
4551 subiw(Rj, Rj, 1);
4552 unroll_2(Rj, &MontgomeryMultiplyGenerator::step);
4553 } block_comment(" } // j");
4554
4555 post2(Ri, Rlen);
4556 addiw(Ri, Ri, 1);
4557 slli(t0, Rlen, 1);
4558 blt(Ri, t0, loop);
4559 bind(end);
4560 }
4561 block_comment("} // i");
4562
4563 normalize(Rlen);
4564
4565 mv(Ra, Pm_base); // Save Pm_base in Ra
4566 restore_regs(); // Restore caller's Pm_base
4567
4568 // Copy our result into caller's Pm_base
4569 reverse(Pm_base, Ra, Rlen, Ri, Rj);
4570
4571 leave();
4572 bind(nothing);
4573 ret();
4574
4575 return entry;
4576 }
4577
4578 /**
4579 *
4580 * Arguments:
4581 *
4582 * Inputs:
4583 * c_rarg0 - int array elements a
4584 * c_rarg1 - int array elements n (the modulus)
4585 * c_rarg2 - int length
4586 * c_rarg3 - int inv
4587 * c_rarg4 - int array elements m (the result)
4588 *
4589 */
4590 address generate_square() {
4591 Label argh;
4592 bind(argh);
4593 stop("MontgomeryMultiply total_allocation must be <= 8192");
4594
4595 align(CodeEntryAlignment);
4596 address entry = pc();
4597
4598 enter();
4599
4600 // Make room.
4601 mv(Ra, 512);
4602 bgt(Rlen, Ra, argh);
4603 slli(Ra, Rlen, exact_log2(4 * sizeof(jint)));
4604 sub(Ra, sp, Ra);
4605 andi(sp, Ra, -2 * wordSize);
4606
4607 srliw(Rlen, Rlen, 1); // length in longwords = len/2
4608
4609 {
4610 // Copy input args, reversing as we go. We use Ra as a
4611 // temporary variable.
4612 reverse(Ra, Pa_base, Rlen, Ri, Rj);
4613 reverse(Ra, Pn_base, Rlen, Ri, Rj);
4614 }
4615
4616 // Push all call-saved registers and also Pm_base which we'll need
4617 // at the end.
4618 save_regs();
4619
4620 mv(Pm_base, Ra);
4621
4622 mv(tmp0, zr);
4623 mv(tmp1, zr);
4624 mv(tmp2, zr);
4625
4626 block_comment("for (int i = 0; i < len; i++) {");
4627 mv(Ri, zr); {
4628 Label loop, end;
4629 bind(loop);
4630 bge(Ri, Rlen, end);
4631
4632 pre1(Ri);
4633
4634 block_comment("for (j = (i+1)/2; j; j--) {"); {
4635 addi(Rj, Ri, 1);
4636 srliw(Rj, Rj, 1);
4637 unroll_2(Rj, &MontgomeryMultiplyGenerator::step_squaring);
4638 } block_comment(" } // j");
4639
4640 last_squaring(Ri);
4641
4642 block_comment(" for (j = i/2; j; j--) {"); {
4643 srliw(Rj, Ri, 1);
4644 unroll_2(Rj, &MontgomeryMultiplyGenerator::extra_step_squaring);
4645 } block_comment(" } // j");
4646
4647 post1_squaring();
4648 addi(Ri, Ri, 1);
4649 blt(Ri, Rlen, loop);
4650
4651 bind(end);
4652 block_comment("} // i");
4653 }
4654
4655 block_comment("for (int i = len; i < 2*len; i++) {");
4656 mv(Ri, Rlen); {
4657 Label loop, end;
4658 bind(loop);
4659 slli(t0, Rlen, 1);
4660 bge(Ri, t0, end);
4661
4662 pre2(Ri, Rlen);
4663
4664 block_comment(" for (j = (2*len-i-1)/2; j; j--) {"); {
4665 slli(Rj, Rlen, 1);
4666 sub(Rj, Rj, Ri);
4667 subi(Rj, Rj, 1);
4668 srliw(Rj, Rj, 1);
4669 unroll_2(Rj, &MontgomeryMultiplyGenerator::step_squaring);
4670 } block_comment(" } // j");
4671
4672 last_squaring(Ri);
4673
4674 block_comment(" for (j = (2*len-i)/2; j; j--) {"); {
4675 slli(Rj, Rlen, 1);
4676 sub(Rj, Rj, Ri);
4677 srliw(Rj, Rj, 1);
4678 unroll_2(Rj, &MontgomeryMultiplyGenerator::extra_step_squaring);
4679 } block_comment(" } // j");
4680
4681 post2(Ri, Rlen);
4682 addi(Ri, Ri, 1);
4683 slli(t0, Rlen, 1);
4684 blt(Ri, t0, loop);
4685
4686 bind(end);
4687 block_comment("} // i");
4688 }
4689
4690 normalize(Rlen);
4691
4692 mv(Ra, Pm_base); // Save Pm_base in Ra
4693 restore_regs(); // Restore caller's Pm_base
4694
4695 // Copy our result into caller's Pm_base
4696 reverse(Pm_base, Ra, Rlen, Ri, Rj);
4697
4698 leave();
4699 ret();
4700
4701 return entry;
4702 }
4703 };
4704
4705 #endif // COMPILER2
4706
4707 address generate_cont_thaw(Continuation::thaw_kind kind) {
4708 bool return_barrier = Continuation::is_thaw_return_barrier(kind);
4709 bool return_barrier_exception = Continuation::is_thaw_return_barrier_exception(kind);
4710
4711 address start = __ pc();
4712
4713 if (return_barrier) {
4714 __ ld(sp, Address(xthread, JavaThread::cont_entry_offset()));
4715 }
4716
4717 #ifndef PRODUCT
4718 {
4719 Label OK;
4720 __ ld(t0, Address(xthread, JavaThread::cont_entry_offset()));
4721 __ beq(sp, t0, OK);
4722 __ stop("incorrect sp");
4723 __ bind(OK);
4724 }
4725 #endif
4726
4727 if (return_barrier) {
4728 // preserve possible return value from a method returning to the return barrier
4729 __ subi(sp, sp, 2 * wordSize);
4730 __ fsd(f10, Address(sp, 0 * wordSize));
4731 __ sd(x10, Address(sp, 1 * wordSize));
4732 }
4733
4734 __ mv(c_rarg1, (return_barrier ? 1 : 0));
4735 __ call_VM_leaf(CAST_FROM_FN_PTR(address, Continuation::prepare_thaw), xthread, c_rarg1);
4736 __ mv(t1, x10); // x10 contains the size of the frames to thaw, 0 if overflow or no more frames
4737
4738 if (return_barrier) {
4739 // restore return value (no safepoint in the call to thaw, so even an oop return value should be OK)
4740 __ ld(x10, Address(sp, 1 * wordSize));
4741 __ fld(f10, Address(sp, 0 * wordSize));
4742 __ addi(sp, sp, 2 * wordSize);
4743 }
4744
4745 #ifndef PRODUCT
4746 {
4747 Label OK;
4748 __ ld(t0, Address(xthread, JavaThread::cont_entry_offset()));
4749 __ beq(sp, t0, OK);
4750 __ stop("incorrect sp");
4751 __ bind(OK);
4752 }
4753 #endif
4754
4755 Label thaw_success;
4756 // t1 contains the size of the frames to thaw, 0 if overflow or no more frames
4757 __ bnez(t1, thaw_success);
4758 __ j(RuntimeAddress(SharedRuntime::throw_StackOverflowError_entry()));
4759 __ bind(thaw_success);
4760
4761 // make room for the thawed frames
4762 __ sub(t0, sp, t1);
4763 __ andi(sp, t0, -16); // align
4764
4765 if (return_barrier) {
4766 // save original return value -- again
4767 __ subi(sp, sp, 2 * wordSize);
4768 __ fsd(f10, Address(sp, 0 * wordSize));
4769 __ sd(x10, Address(sp, 1 * wordSize));
4770 }
4771
4772 // If we want, we can templatize thaw by kind, and have three different entries
4773 __ mv(c_rarg1, kind);
4774
4775 __ call_VM_leaf(Continuation::thaw_entry(), xthread, c_rarg1);
4776 __ mv(t1, x10); // x10 is the sp of the yielding frame
4777
4778 if (return_barrier) {
4779 // restore return value (no safepoint in the call to thaw, so even an oop return value should be OK)
4780 __ ld(x10, Address(sp, 1 * wordSize));
4781 __ fld(f10, Address(sp, 0 * wordSize));
4782 __ addi(sp, sp, 2 * wordSize);
4783 } else {
4784 __ mv(x10, zr); // return 0 (success) from doYield
4785 }
4786
4787 // we're now on the yield frame (which is in an address above us b/c sp has been pushed down)
4788 __ mv(fp, t1);
4789 __ subi(sp, t1, 2 * wordSize); // now pointing to fp spill
4790
4791 if (return_barrier_exception) {
4792 __ ld(c_rarg1, Address(fp, -1 * wordSize)); // return address
4793 __ verify_oop(x10);
4794 __ mv(x9, x10); // save return value contaning the exception oop in callee-saved x9
4795
4796 __ call_VM_leaf(CAST_FROM_FN_PTR(address, SharedRuntime::exception_handler_for_return_address), xthread, c_rarg1);
4797
4798 // see OptoRuntime::generate_exception_blob: x10 -- exception oop, x13 -- exception pc
4799
4800 __ mv(x11, x10); // the exception handler
4801 __ mv(x10, x9); // restore return value contaning the exception oop
4802 __ verify_oop(x10);
4803
4804 __ leave();
4805 __ mv(x13, ra);
4806 __ jr(x11); // the exception handler
4807 } else {
4808 // We're "returning" into the topmost thawed frame; see Thaw::push_return_frame
4809 __ leave();
4810 __ ret();
4811 }
4812
4813 return start;
4814 }
4815
4816 address generate_cont_thaw() {
4817 if (!Continuations::enabled()) return nullptr;
4818
4819 StubId stub_id = StubId::stubgen_cont_thaw_id;
4820 StubCodeMark mark(this, stub_id);
4821 address start = __ pc();
4822 generate_cont_thaw(Continuation::thaw_top);
4823 return start;
4824 }
4825
4826 address generate_cont_returnBarrier() {
4827 if (!Continuations::enabled()) return nullptr;
4828
4829 // TODO: will probably need multiple return barriers depending on return type
4830 StubId stub_id = StubId::stubgen_cont_returnBarrier_id;
4831 StubCodeMark mark(this, stub_id);
4832 address start = __ pc();
4833
4834 generate_cont_thaw(Continuation::thaw_return_barrier);
4835
4836 return start;
4837 }
4838
4839 address generate_cont_returnBarrier_exception() {
4840 if (!Continuations::enabled()) return nullptr;
4841
4842 StubId stub_id = StubId::stubgen_cont_returnBarrierExc_id;
4843 StubCodeMark mark(this, stub_id);
4844 address start = __ pc();
4845
4846 generate_cont_thaw(Continuation::thaw_return_barrier_exception);
4847
4848 return start;
4849 }
4850
4851 address generate_cont_preempt_stub() {
4852 if (!Continuations::enabled()) return nullptr;
4853 StubId stub_id = StubId::stubgen_cont_preempt_id;
4854 StubCodeMark mark(this, stub_id);
4855 address start = __ pc();
4856
4857 __ reset_last_Java_frame(true);
4858
4859 // Set sp to enterSpecial frame, i.e. remove all frames copied into the heap.
4860 __ ld(sp, Address(xthread, JavaThread::cont_entry_offset()));
4861
4862 Label preemption_cancelled;
4863 __ lbu(t0, Address(xthread, JavaThread::preemption_cancelled_offset()));
4864 __ bnez(t0, preemption_cancelled);
4865
4866 // Remove enterSpecial frame from the stack and return to Continuation.run() to unmount.
4867 SharedRuntime::continuation_enter_cleanup(_masm);
4868 __ leave();
4869 __ ret();
4870
4871 // We acquired the monitor after freezing the frames so call thaw to continue execution.
4872 __ bind(preemption_cancelled);
4873 __ sb(zr, Address(xthread, JavaThread::preemption_cancelled_offset()));
4874 __ la(fp, Address(sp, checked_cast<int32_t>(ContinuationEntry::size() + 2 * wordSize)));
4875 __ la(t1, ExternalAddress(ContinuationEntry::thaw_call_pc_address()));
4876 __ ld(t1, Address(t1));
4877 __ jr(t1);
4878
4879 return start;
4880 }
4881
4882 #ifdef COMPILER2
4883
4884 #undef __
4885 #define __ this->
4886
4887 class Sha2Generator : public MacroAssembler {
4888 StubCodeGenerator* _cgen;
4889 public:
4890 Sha2Generator(MacroAssembler* masm, StubCodeGenerator* cgen) : MacroAssembler(masm->code()), _cgen(cgen) {}
4891 address generate_sha256_implCompress(StubId stub_id) {
4892 return generate_sha2_implCompress(Assembler::e32, stub_id);
4893 }
4894 address generate_sha512_implCompress(StubId stub_id) {
4895 return generate_sha2_implCompress(Assembler::e64, stub_id);
4896 }
4897 private:
4898
4899 void vleXX_v(Assembler::SEW vset_sew, VectorRegister vr, Register sr) {
4900 if (vset_sew == Assembler::e32) __ vle32_v(vr, sr);
4901 else __ vle64_v(vr, sr);
4902 }
4903
4904 void vseXX_v(Assembler::SEW vset_sew, VectorRegister vr, Register sr) {
4905 if (vset_sew == Assembler::e32) __ vse32_v(vr, sr);
4906 else __ vse64_v(vr, sr);
4907 }
4908
4909 // Overview of the logic in each "quad round".
4910 //
4911 // The code below repeats 16/20 times the logic implementing four rounds
4912 // of the SHA-256/512 core loop as documented by NIST. 16/20 "quad rounds"
4913 // to implementing the 64/80 single rounds.
4914 //
4915 // // Load four word (u32/64) constants (K[t+3], K[t+2], K[t+1], K[t+0])
4916 // // Output:
4917 // // vTmp1 = {K[t+3], K[t+2], K[t+1], K[t+0]}
4918 // vl1reXX.v vTmp1, ofs
4919 //
4920 // // Increment word constant address by stride (16/32 bytes, 4*4B/8B, 128b/256b)
4921 // addi ofs, ofs, 16/32
4922 //
4923 // // Add constants to message schedule words:
4924 // // Input
4925 // // vTmp1 = {K[t+3], K[t+2], K[t+1], K[t+0]}
4926 // // vW0 = {W[t+3], W[t+2], W[t+1], W[t+0]}; // Vt0 = W[3:0];
4927 // // Output
4928 // // vTmp0 = {W[t+3]+K[t+3], W[t+2]+K[t+2], W[t+1]+K[t+1], W[t+0]+K[t+0]}
4929 // vadd.vv vTmp0, vTmp1, vW0
4930 //
4931 // // 2 rounds of working variables updates.
4932 // // vState1[t+4] <- vState1[t], vState0[t], vTmp0[t]
4933 // // Input:
4934 // // vState1 = {c[t],d[t],g[t],h[t]} " = vState1[t] "
4935 // // vState0 = {a[t],b[t],e[t],f[t]}
4936 // // vTmp0 = {W[t+3]+K[t+3], W[t+2]+K[t+2], W[t+1]+K[t+1], W[t+0]+K[t+0]}
4937 // // Output:
4938 // // vState1 = {f[t+2],e[t+2],b[t+2],a[t+2]} " = vState0[t+2] "
4939 // // = {h[t+4],g[t+4],d[t+4],c[t+4]} " = vState1[t+4] "
4940 // vsha2cl.vv vState1, vState0, vTmp0
4941 //
4942 // // 2 rounds of working variables updates.
4943 // // vState0[t+4] <- vState0[t], vState0[t+2], vTmp0[t]
4944 // // Input
4945 // // vState0 = {a[t],b[t],e[t],f[t]} " = vState0[t] "
4946 // // = {h[t+2],g[t+2],d[t+2],c[t+2]} " = vState1[t+2] "
4947 // // vState1 = {f[t+2],e[t+2],b[t+2],a[t+2]} " = vState0[t+2] "
4948 // // vTmp0 = {W[t+3]+K[t+3], W[t+2]+K[t+2], W[t+1]+K[t+1], W[t+0]+K[t+0]}
4949 // // Output:
4950 // // vState0 = {f[t+4],e[t+4],b[t+4],a[t+4]} " = vState0[t+4] "
4951 // vsha2ch.vv vState0, vState1, vTmp0
4952 //
4953 // // Combine 2QW into 1QW
4954 // //
4955 // // To generate the next 4 words, "new_vW0"/"vTmp0" from vW0-vW3, vsha2ms needs
4956 // // vW0[0..3], vW1[0], vW2[1..3], vW3[0, 2..3]
4957 // // and it can only take 3 vectors as inputs. Hence we need to combine
4958 // // vW1[0] and vW2[1..3] in a single vector.
4959 // //
4960 // // vmerge Vt4, Vt1, Vt2, V0
4961 // // Input
4962 // // V0 = mask // first word from vW2, 1..3 words from vW1
4963 // // vW2 = {Wt-8, Wt-7, Wt-6, Wt-5}
4964 // // vW1 = {Wt-12, Wt-11, Wt-10, Wt-9}
4965 // // Output
4966 // // Vt4 = {Wt-12, Wt-7, Wt-6, Wt-5}
4967 // vmerge.vvm vTmp0, vW2, vW1, v0
4968 //
4969 // // Generate next Four Message Schedule Words (hence allowing for 4 more rounds)
4970 // // Input
4971 // // vW0 = {W[t+ 3], W[t+ 2], W[t+ 1], W[t+ 0]} W[ 3: 0]
4972 // // vW3 = {W[t+15], W[t+14], W[t+13], W[t+12]} W[15:12]
4973 // // vTmp0 = {W[t+11], W[t+10], W[t+ 9], W[t+ 4]} W[11: 9,4]
4974 // // Output (next four message schedule words)
4975 // // vW0 = {W[t+19], W[t+18], W[t+17], W[t+16]} W[19:16]
4976 // vsha2ms.vv vW0, vTmp0, vW3
4977 //
4978 // BEFORE
4979 // vW0 - vW3 hold the message schedule words (initially the block words)
4980 // vW0 = W[ 3: 0] "oldest"
4981 // vW1 = W[ 7: 4]
4982 // vW2 = W[11: 8]
4983 // vW3 = W[15:12] "newest"
4984 //
4985 // vt6 - vt7 hold the working state variables
4986 // vState0 = {a[t],b[t],e[t],f[t]} // initially {H5,H4,H1,H0}
4987 // vState1 = {c[t],d[t],g[t],h[t]} // initially {H7,H6,H3,H2}
4988 //
4989 // AFTER
4990 // vW0 - vW3 hold the message schedule words (initially the block words)
4991 // vW1 = W[ 7: 4] "oldest"
4992 // vW2 = W[11: 8]
4993 // vW3 = W[15:12]
4994 // vW0 = W[19:16] "newest"
4995 //
4996 // vState0 and vState1 hold the working state variables
4997 // vState0 = {a[t+4],b[t+4],e[t+4],f[t+4]}
4998 // vState1 = {c[t+4],d[t+4],g[t+4],h[t+4]}
4999 //
5000 // The group of vectors vW0,vW1,vW2,vW3 is "rotated" by one in each quad-round,
5001 // hence the uses of those vectors rotate in each round, and we get back to the
5002 // initial configuration every 4 quad-rounds. We could avoid those changes at
5003 // the cost of moving those vectors at the end of each quad-rounds.
5004 void sha2_quad_round(Assembler::SEW vset_sew, VectorRegister rot1, VectorRegister rot2, VectorRegister rot3, VectorRegister rot4,
5005 Register scalarconst, VectorRegister vtemp, VectorRegister vtemp2, VectorRegister v_abef, VectorRegister v_cdgh,
5006 bool gen_words = true, bool step_const = true) {
5007 __ vleXX_v(vset_sew, vtemp, scalarconst);
5008 if (step_const) {
5009 __ addi(scalarconst, scalarconst, vset_sew == Assembler::e32 ? 16 : 32);
5010 }
5011 __ vadd_vv(vtemp2, vtemp, rot1);
5012 __ vsha2cl_vv(v_cdgh, v_abef, vtemp2);
5013 __ vsha2ch_vv(v_abef, v_cdgh, vtemp2);
5014 if (gen_words) {
5015 __ vmerge_vvm(vtemp2, rot3, rot2);
5016 __ vsha2ms_vv(rot1, vtemp2, rot4);
5017 }
5018 }
5019
5020 // Arguments:
5021 //
5022 // Inputs:
5023 // c_rarg0 - byte[] source+offset
5024 // c_rarg1 - int[] SHA.state
5025 // c_rarg2 - int offset
5026 // c_rarg3 - int limit
5027 //
5028 address generate_sha2_implCompress(Assembler::SEW vset_sew, StubId stub_id) {
5029 alignas(64) static const uint32_t round_consts_256[64] = {
5030 0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5,
5031 0x3956c25b, 0x59f111f1, 0x923f82a4, 0xab1c5ed5,
5032 0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3,
5033 0x72be5d74, 0x80deb1fe, 0x9bdc06a7, 0xc19bf174,
5034 0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc,
5035 0x2de92c6f, 0x4a7484aa, 0x5cb0a9dc, 0x76f988da,
5036 0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7,
5037 0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967,
5038 0x27b70a85, 0x2e1b2138, 0x4d2c6dfc, 0x53380d13,
5039 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85,
5040 0xa2bfe8a1, 0xa81a664b, 0xc24b8b70, 0xc76c51a3,
5041 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070,
5042 0x19a4c116, 0x1e376c08, 0x2748774c, 0x34b0bcb5,
5043 0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3,
5044 0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208,
5045 0x90befffa, 0xa4506ceb, 0xbef9a3f7, 0xc67178f2,
5046 };
5047 alignas(64) static const uint64_t round_consts_512[80] = {
5048 0x428a2f98d728ae22l, 0x7137449123ef65cdl, 0xb5c0fbcfec4d3b2fl,
5049 0xe9b5dba58189dbbcl, 0x3956c25bf348b538l, 0x59f111f1b605d019l,
5050 0x923f82a4af194f9bl, 0xab1c5ed5da6d8118l, 0xd807aa98a3030242l,
5051 0x12835b0145706fbel, 0x243185be4ee4b28cl, 0x550c7dc3d5ffb4e2l,
5052 0x72be5d74f27b896fl, 0x80deb1fe3b1696b1l, 0x9bdc06a725c71235l,
5053 0xc19bf174cf692694l, 0xe49b69c19ef14ad2l, 0xefbe4786384f25e3l,
5054 0x0fc19dc68b8cd5b5l, 0x240ca1cc77ac9c65l, 0x2de92c6f592b0275l,
5055 0x4a7484aa6ea6e483l, 0x5cb0a9dcbd41fbd4l, 0x76f988da831153b5l,
5056 0x983e5152ee66dfabl, 0xa831c66d2db43210l, 0xb00327c898fb213fl,
5057 0xbf597fc7beef0ee4l, 0xc6e00bf33da88fc2l, 0xd5a79147930aa725l,
5058 0x06ca6351e003826fl, 0x142929670a0e6e70l, 0x27b70a8546d22ffcl,
5059 0x2e1b21385c26c926l, 0x4d2c6dfc5ac42aedl, 0x53380d139d95b3dfl,
5060 0x650a73548baf63del, 0x766a0abb3c77b2a8l, 0x81c2c92e47edaee6l,
5061 0x92722c851482353bl, 0xa2bfe8a14cf10364l, 0xa81a664bbc423001l,
5062 0xc24b8b70d0f89791l, 0xc76c51a30654be30l, 0xd192e819d6ef5218l,
5063 0xd69906245565a910l, 0xf40e35855771202al, 0x106aa07032bbd1b8l,
5064 0x19a4c116b8d2d0c8l, 0x1e376c085141ab53l, 0x2748774cdf8eeb99l,
5065 0x34b0bcb5e19b48a8l, 0x391c0cb3c5c95a63l, 0x4ed8aa4ae3418acbl,
5066 0x5b9cca4f7763e373l, 0x682e6ff3d6b2b8a3l, 0x748f82ee5defb2fcl,
5067 0x78a5636f43172f60l, 0x84c87814a1f0ab72l, 0x8cc702081a6439ecl,
5068 0x90befffa23631e28l, 0xa4506cebde82bde9l, 0xbef9a3f7b2c67915l,
5069 0xc67178f2e372532bl, 0xca273eceea26619cl, 0xd186b8c721c0c207l,
5070 0xeada7dd6cde0eb1el, 0xf57d4f7fee6ed178l, 0x06f067aa72176fbal,
5071 0x0a637dc5a2c898a6l, 0x113f9804bef90dael, 0x1b710b35131c471bl,
5072 0x28db77f523047d84l, 0x32caab7b40c72493l, 0x3c9ebe0a15c9bebcl,
5073 0x431d67c49c100d4cl, 0x4cc5d4becb3e42b6l, 0x597f299cfc657e2al,
5074 0x5fcb6fab3ad6faecl, 0x6c44198c4a475817l
5075 };
5076 const int const_add = vset_sew == Assembler::e32 ? 16 : 32;
5077
5078 bool multi_block;
5079 switch (stub_id) {
5080 case StubId::stubgen_sha256_implCompress_id:
5081 assert (vset_sew == Assembler::e32, "wrong macroassembler for stub");
5082 multi_block = false;
5083 break;
5084 case StubId::stubgen_sha256_implCompressMB_id:
5085 assert (vset_sew == Assembler::e32, "wrong macroassembler for stub");
5086 multi_block = true;
5087 break;
5088 case StubId::stubgen_sha512_implCompress_id:
5089 assert (vset_sew == Assembler::e64, "wrong macroassembler for stub");
5090 multi_block = false;
5091 break;
5092 case StubId::stubgen_sha512_implCompressMB_id:
5093 assert (vset_sew == Assembler::e64, "wrong macroassembler for stub");
5094 multi_block = true;
5095 break;
5096 default:
5097 ShouldNotReachHere();
5098 };
5099 __ align(CodeEntryAlignment);
5100 StubCodeMark mark(_cgen, stub_id);
5101 address start = __ pc();
5102
5103 Register buf = c_rarg0;
5104 Register state = c_rarg1;
5105 Register ofs = c_rarg2;
5106 Register limit = c_rarg3;
5107 Register consts = t2; // caller saved
5108 Register state_c = x28; // caller saved
5109 VectorRegister vindex = v2;
5110 VectorRegister vW0 = v4;
5111 VectorRegister vW1 = v6;
5112 VectorRegister vW2 = v8;
5113 VectorRegister vW3 = v10;
5114 VectorRegister vState0 = v12;
5115 VectorRegister vState1 = v14;
5116 VectorRegister vHash0 = v16;
5117 VectorRegister vHash1 = v18;
5118 VectorRegister vTmp0 = v20;
5119 VectorRegister vTmp1 = v22;
5120
5121 Label multi_block_loop;
5122
5123 __ enter();
5124
5125 address constant_table = vset_sew == Assembler::e32 ? (address)round_consts_256 : (address)round_consts_512;
5126 la(consts, ExternalAddress(constant_table));
5127
5128 // Register use in this function:
5129 //
5130 // VECTORS
5131 // vW0 - vW3 (512/1024-bits / 4*128/256 bits / 4*4*32/65 bits), hold the message
5132 // schedule words (Wt). They start with the message block
5133 // content (W0 to W15), then further words in the message
5134 // schedule generated via vsha2ms from previous Wt.
5135 // Initially:
5136 // vW0 = W[ 3:0] = { W3, W2, W1, W0}
5137 // vW1 = W[ 7:4] = { W7, W6, W5, W4}
5138 // vW2 = W[ 11:8] = {W11, W10, W9, W8}
5139 // vW3 = W[15:12] = {W15, W14, W13, W12}
5140 //
5141 // vState0 - vState1 hold the working state variables (a, b, ..., h)
5142 // vState0 = {f[t],e[t],b[t],a[t]}
5143 // vState1 = {h[t],g[t],d[t],c[t]}
5144 // Initially:
5145 // vState0 = {H5i-1, H4i-1, H1i-1 , H0i-1}
5146 // vState1 = {H7i-i, H6i-1, H3i-1 , H2i-1}
5147 //
5148 // v0 = masks for vrgather/vmerge. Single value during the 16 rounds.
5149 //
5150 // vTmp0 = temporary, Wt+Kt
5151 // vTmp1 = temporary, Kt
5152 //
5153 // vHash0/vHash1 = hold the initial values of the hash, byte-swapped.
5154 //
5155 // During most of the function the vector state is configured so that each
5156 // vector is interpreted as containing four 32/64 bits (e32/e64) elements (128/256 bits).
5157
5158 // vsha2ch/vsha2cl uses EGW of 4*SEW.
5159 // SHA256 SEW = e32, EGW = 128-bits
5160 // SHA512 SEW = e64, EGW = 256-bits
5161 //
5162 // VLEN is required to be at least 128.
5163 // For the case of VLEN=128 and SHA512 we need LMUL=2 to work with 4*e64 (EGW = 256)
5164 //
5165 // m1: LMUL=1/2
5166 // ta: tail agnostic (don't care about those lanes)
5167 // ma: mask agnostic (don't care about those lanes)
5168 // x0 is not written, we known the number of vector elements.
5169
5170 if (vset_sew == Assembler::e64 && MaxVectorSize == 16) { // SHA512 and VLEN = 128
5171 __ vsetivli(x0, 4, vset_sew, Assembler::m2, Assembler::ma, Assembler::ta);
5172 } else {
5173 __ vsetivli(x0, 4, vset_sew, Assembler::m1, Assembler::ma, Assembler::ta);
5174 }
5175
5176 int64_t indexes = vset_sew == Assembler::e32 ? 0x00041014ul : 0x00082028ul;
5177 __ li(t0, indexes);
5178 __ vmv_v_x(vindex, t0);
5179
5180 // Step-over a,b, so we are pointing to c.
5181 // const_add is equal to 4x state variable, div by 2 is thus 2, a,b
5182 __ addi(state_c, state, const_add/2);
5183
5184 // Use index-load to get {f,e,b,a},{h,g,d,c}
5185 __ vluxei8_v(vState0, state, vindex);
5186 __ vluxei8_v(vState1, state_c, vindex);
5187
5188 __ bind(multi_block_loop);
5189
5190 // Capture the initial H values in vHash0 and vHash1 to allow for computing
5191 // the resulting H', since H' = H+{a',b',c',...,h'}.
5192 __ vmv_v_v(vHash0, vState0);
5193 __ vmv_v_v(vHash1, vState1);
5194
5195 // Load the 512/1024-bits of the message block in vW0-vW3 and perform
5196 // an endian swap on each 4/8 bytes element.
5197 //
5198 // If Zvkb is not implemented one can use vrgather
5199 // with an index sequence to byte-swap.
5200 // sequence = [3 2 1 0 7 6 5 4 11 10 9 8 15 14 13 12]
5201 // <https://oeis.org/A004444> gives us "N ^ 3" as a nice formula to generate
5202 // this sequence. 'vid' gives us the N.
5203 __ vleXX_v(vset_sew, vW0, buf);
5204 __ vrev8_v(vW0, vW0);
5205 __ addi(buf, buf, const_add);
5206 __ vleXX_v(vset_sew, vW1, buf);
5207 __ vrev8_v(vW1, vW1);
5208 __ addi(buf, buf, const_add);
5209 __ vleXX_v(vset_sew, vW2, buf);
5210 __ vrev8_v(vW2, vW2);
5211 __ addi(buf, buf, const_add);
5212 __ vleXX_v(vset_sew, vW3, buf);
5213 __ vrev8_v(vW3, vW3);
5214 __ addi(buf, buf, const_add);
5215
5216 // Set v0 up for the vmerge that replaces the first word (idx==0)
5217 __ vid_v(v0);
5218 __ vmseq_vi(v0, v0, 0x0); // v0.mask[i] = (i == 0 ? 1 : 0)
5219
5220 VectorRegister rotation_regs[] = {vW0, vW1, vW2, vW3};
5221 int rot_pos = 0;
5222 // Quad-round #0 (+0, vW0->vW1->vW2->vW3) ... #11 (+3, vW3->vW0->vW1->vW2)
5223 const int qr_end = vset_sew == Assembler::e32 ? 12 : 16;
5224 for (int i = 0; i < qr_end; i++) {
5225 sha2_quad_round(vset_sew,
5226 rotation_regs[(rot_pos + 0) & 0x3],
5227 rotation_regs[(rot_pos + 1) & 0x3],
5228 rotation_regs[(rot_pos + 2) & 0x3],
5229 rotation_regs[(rot_pos + 3) & 0x3],
5230 consts,
5231 vTmp1, vTmp0, vState0, vState1);
5232 ++rot_pos;
5233 }
5234 // Quad-round #12 (+0, vW0->vW1->vW2->vW3) ... #15 (+3, vW3->vW0->vW1->vW2)
5235 // Note that we stop generating new message schedule words (Wt, vW0-13)
5236 // as we already generated all the words we end up consuming (i.e., W[63:60]).
5237 const int qr_c_end = qr_end + 4;
5238 for (int i = qr_end; i < qr_c_end; i++) {
5239 sha2_quad_round(vset_sew,
5240 rotation_regs[(rot_pos + 0) & 0x3],
5241 rotation_regs[(rot_pos + 1) & 0x3],
5242 rotation_regs[(rot_pos + 2) & 0x3],
5243 rotation_regs[(rot_pos + 3) & 0x3],
5244 consts,
5245 vTmp1, vTmp0, vState0, vState1, false, i < (qr_c_end-1));
5246 ++rot_pos;
5247 }
5248
5249 //--------------------------------------------------------------------------------
5250 // Compute the updated hash value H'
5251 // H' = H + {h',g',...,b',a'}
5252 // = {h,g,...,b,a} + {h',g',...,b',a'}
5253 // = {h+h',g+g',...,b+b',a+a'}
5254
5255 // H' = H+{a',b',c',...,h'}
5256 __ vadd_vv(vState0, vHash0, vState0);
5257 __ vadd_vv(vState1, vHash1, vState1);
5258
5259 if (multi_block) {
5260 int total_adds = vset_sew == Assembler::e32 ? 240 : 608;
5261 __ subi(consts, consts, total_adds);
5262 __ addi(ofs, ofs, vset_sew == Assembler::e32 ? 64 : 128);
5263 __ ble(ofs, limit, multi_block_loop);
5264 __ mv(c_rarg0, ofs); // return ofs
5265 }
5266
5267 // Store H[0..8] = {a,b,c,d,e,f,g,h} from
5268 // vState0 = {f,e,b,a}
5269 // vState1 = {h,g,d,c}
5270 __ vsuxei8_v(vState0, state, vindex);
5271 __ vsuxei8_v(vState1, state_c, vindex);
5272
5273 __ leave();
5274 __ ret();
5275
5276 return start;
5277 }
5278 };
5279
5280 #undef __
5281 #define __ _masm->
5282
5283 // Set of L registers that correspond to a contiguous memory area.
5284 // Each 64-bit register typically corresponds to 2 32-bit integers.
5285 template <uint L>
5286 class RegCache {
5287 private:
5288 MacroAssembler *_masm;
5289 Register _regs[L];
5290
5291 public:
5292 RegCache(MacroAssembler *masm, RegSet rs): _masm(masm) {
5293 assert(rs.size() == L, "%u registers are used to cache %u 4-byte data", rs.size(), 2 * L);
5294 auto it = rs.begin();
5295 for (auto &r: _regs) {
5296 r = *it;
5297 ++it;
5298 }
5299 }
5300
5301 // generate load for the i'th register
5302 void gen_load(uint i, Register base) {
5303 assert(i < L, "invalid i: %u", i);
5304 __ ld(_regs[i], Address(base, 8 * i));
5305 }
5306
5307 // add i'th 32-bit integer to dest
5308 void add_u32(const Register dest, uint i, const Register rtmp = t0) {
5309 assert(i < 2 * L, "invalid i: %u", i);
5310
5311 if (is_even(i)) {
5312 // Use the bottom 32 bits. No need to mask off the top 32 bits
5313 // as addw will do the right thing.
5314 __ addw(dest, dest, _regs[i / 2]);
5315 } else {
5316 // Use the top 32 bits by right-shifting them.
5317 __ srli(rtmp, _regs[i / 2], 32);
5318 __ addw(dest, dest, rtmp);
5319 }
5320 }
5321 };
5322
5323 typedef RegCache<8> BufRegCache;
5324
5325 // a += value + x + ac;
5326 // a = Integer.rotateLeft(a, s) + b;
5327 void m5_FF_GG_HH_II_epilogue(BufRegCache& reg_cache,
5328 Register a, Register b, Register c, Register d,
5329 int k, int s, int t,
5330 Register value) {
5331 // a += ac
5332 __ addw(a, a, t, t1);
5333
5334 // a += x;
5335 reg_cache.add_u32(a, k);
5336 // a += value;
5337 __ addw(a, a, value);
5338
5339 // a = Integer.rotateLeft(a, s) + b;
5340 __ rolw(a, a, s);
5341 __ addw(a, a, b);
5342 }
5343
5344 // a += ((b & c) | ((~b) & d)) + x + ac;
5345 // a = Integer.rotateLeft(a, s) + b;
5346 void md5_FF(BufRegCache& reg_cache,
5347 Register a, Register b, Register c, Register d,
5348 int k, int s, int t,
5349 Register rtmp1, Register rtmp2) {
5350 // rtmp1 = b & c
5351 __ andr(rtmp1, b, c);
5352
5353 // rtmp2 = (~b) & d
5354 __ andn(rtmp2, d, b);
5355
5356 // rtmp1 = (b & c) | ((~b) & d)
5357 __ orr(rtmp1, rtmp1, rtmp2);
5358
5359 m5_FF_GG_HH_II_epilogue(reg_cache, a, b, c, d, k, s, t, rtmp1);
5360 }
5361
5362 // a += ((b & d) | (c & (~d))) + x + ac;
5363 // a = Integer.rotateLeft(a, s) + b;
5364 void md5_GG(BufRegCache& reg_cache,
5365 Register a, Register b, Register c, Register d,
5366 int k, int s, int t,
5367 Register rtmp1, Register rtmp2) {
5368 // rtmp1 = b & d
5369 __ andr(rtmp1, b, d);
5370
5371 // rtmp2 = c & (~d)
5372 __ andn(rtmp2, c, d);
5373
5374 // rtmp1 = (b & d) | (c & (~d))
5375 __ orr(rtmp1, rtmp1, rtmp2);
5376
5377 m5_FF_GG_HH_II_epilogue(reg_cache, a, b, c, d, k, s, t, rtmp1);
5378 }
5379
5380 // a += ((b ^ c) ^ d) + x + ac;
5381 // a = Integer.rotateLeft(a, s) + b;
5382 void md5_HH(BufRegCache& reg_cache,
5383 Register a, Register b, Register c, Register d,
5384 int k, int s, int t,
5385 Register rtmp1, Register rtmp2) {
5386 // rtmp1 = (b ^ c) ^ d
5387 __ xorr(rtmp2, b, c);
5388 __ xorr(rtmp1, rtmp2, d);
5389
5390 m5_FF_GG_HH_II_epilogue(reg_cache, a, b, c, d, k, s, t, rtmp1);
5391 }
5392
5393 // a += (c ^ (b | (~d))) + x + ac;
5394 // a = Integer.rotateLeft(a, s) + b;
5395 void md5_II(BufRegCache& reg_cache,
5396 Register a, Register b, Register c, Register d,
5397 int k, int s, int t,
5398 Register rtmp1, Register rtmp2) {
5399 // rtmp1 = c ^ (b | (~d))
5400 __ orn(rtmp2, b, d);
5401 __ xorr(rtmp1, c, rtmp2);
5402
5403 m5_FF_GG_HH_II_epilogue(reg_cache, a, b, c, d, k, s, t, rtmp1);
5404 }
5405
5406 // Arguments:
5407 //
5408 // Inputs:
5409 // c_rarg0 - byte[] source+offset
5410 // c_rarg1 - int[] SHA.state
5411 // c_rarg2 - int offset (multi_block == True)
5412 // c_rarg3 - int limit (multi_block == True)
5413 //
5414 // Registers:
5415 // x0 zero (zero)
5416 // x1 ra (return address)
5417 // x2 sp (stack pointer)
5418 // x3 gp (global pointer)
5419 // x4 tp (thread pointer)
5420 // x5 t0 (tmp register)
5421 // x6 t1 (tmp register)
5422 // x7 t2 state0
5423 // x8 f0/s0 (frame pointer)
5424 // x9 s1
5425 // x10 a0 rtmp1 / c_rarg0
5426 // x11 a1 rtmp2 / c_rarg1
5427 // x12 a2 a / c_rarg2
5428 // x13 a3 b / c_rarg3
5429 // x14 a4 c
5430 // x15 a5 d
5431 // x16 a6 buf
5432 // x17 a7 state
5433 // x18 s2 ofs [saved-reg] (multi_block == True)
5434 // x19 s3 limit [saved-reg] (multi_block == True)
5435 // x20 s4 state1 [saved-reg]
5436 // x21 s5 state2 [saved-reg]
5437 // x22 s6 state3 [saved-reg]
5438 // x23 s7
5439 // x24 s8 buf0 [saved-reg]
5440 // x25 s9 buf1 [saved-reg]
5441 // x26 s10 buf2 [saved-reg]
5442 // x27 s11 buf3 [saved-reg]
5443 // x28 t3 buf4
5444 // x29 t4 buf5
5445 // x30 t5 buf6
5446 // x31 t6 buf7
5447 address generate_md5_implCompress(StubId stub_id) {
5448 __ align(CodeEntryAlignment);
5449 bool multi_block;
5450 switch (stub_id) {
5451 case StubId::stubgen_md5_implCompress_id:
5452 multi_block = false;
5453 break;
5454 case StubId::stubgen_md5_implCompressMB_id:
5455 multi_block = true;
5456 break;
5457 default:
5458 ShouldNotReachHere();
5459 };
5460 StubCodeMark mark(this, stub_id);
5461 address start = __ pc();
5462
5463 // rotation constants
5464 const int S11 = 7;
5465 const int S12 = 12;
5466 const int S13 = 17;
5467 const int S14 = 22;
5468 const int S21 = 5;
5469 const int S22 = 9;
5470 const int S23 = 14;
5471 const int S24 = 20;
5472 const int S31 = 4;
5473 const int S32 = 11;
5474 const int S33 = 16;
5475 const int S34 = 23;
5476 const int S41 = 6;
5477 const int S42 = 10;
5478 const int S43 = 15;
5479 const int S44 = 21;
5480
5481 const int64_t mask32 = 0xffffffff;
5482
5483 Register buf_arg = c_rarg0; // a0
5484 Register state_arg = c_rarg1; // a1
5485 Register ofs_arg = c_rarg2; // a2
5486 Register limit_arg = c_rarg3; // a3
5487
5488 // we'll copy the args to these registers to free up a0-a3
5489 // to use for other values manipulated by instructions
5490 // that can be compressed
5491 Register buf = x16; // a6
5492 Register state = x17; // a7
5493 Register ofs = x18; // s2
5494 Register limit = x19; // s3
5495
5496 // using x12->15 to allow compressed instructions
5497 Register a = x12; // a2
5498 Register b = x13; // a3
5499 Register c = x14; // a4
5500 Register d = x15; // a5
5501
5502 Register state0 = x7; // t2
5503 Register state1 = x20; // s4
5504 Register state2 = x21; // s5
5505 Register state3 = x22; // s6
5506
5507 // using x10->x11 to allow compressed instructions
5508 Register rtmp1 = x10; // a0
5509 Register rtmp2 = x11; // a1
5510
5511 RegSet reg_cache_saved_regs = RegSet::of(x24, x25, x26, x27); // s8, s9, s10, s11
5512 RegSet reg_cache_regs;
5513 reg_cache_regs += reg_cache_saved_regs;
5514 reg_cache_regs += RegSet::of(t3, t4, t5, t6);
5515 BufRegCache reg_cache(_masm, reg_cache_regs);
5516
5517 RegSet saved_regs;
5518 if (multi_block) {
5519 saved_regs += RegSet::of(ofs, limit);
5520 }
5521 saved_regs += RegSet::of(state1, state2, state3);
5522 saved_regs += reg_cache_saved_regs;
5523
5524 __ push_reg(saved_regs, sp);
5525
5526 __ mv(buf, buf_arg);
5527 __ mv(state, state_arg);
5528 if (multi_block) {
5529 __ mv(ofs, ofs_arg);
5530 __ mv(limit, limit_arg);
5531 }
5532
5533 // to minimize the number of memory operations:
5534 // read the 4 state 4-byte values in pairs, with a single ld,
5535 // and split them into 2 registers.
5536 //
5537 // And, as the core algorithm of md5 works on 32-bits words, so
5538 // in the following code, it does not care about the content of
5539 // higher 32-bits in state[x]. Based on this observation,
5540 // we can apply further optimization, which is to just ignore the
5541 // higher 32-bits in state0/state2, rather than set the higher
5542 // 32-bits of state0/state2 to zero explicitly with extra instructions.
5543 __ ld(state0, Address(state));
5544 __ srli(state1, state0, 32);
5545 __ ld(state2, Address(state, 8));
5546 __ srli(state3, state2, 32);
5547
5548 Label md5_loop;
5549 __ BIND(md5_loop);
5550
5551 __ mv(a, state0);
5552 __ mv(b, state1);
5553 __ mv(c, state2);
5554 __ mv(d, state3);
5555
5556 // Round 1
5557 reg_cache.gen_load(0, buf);
5558 md5_FF(reg_cache, a, b, c, d, 0, S11, 0xd76aa478, rtmp1, rtmp2);
5559 md5_FF(reg_cache, d, a, b, c, 1, S12, 0xe8c7b756, rtmp1, rtmp2);
5560 reg_cache.gen_load(1, buf);
5561 md5_FF(reg_cache, c, d, a, b, 2, S13, 0x242070db, rtmp1, rtmp2);
5562 md5_FF(reg_cache, b, c, d, a, 3, S14, 0xc1bdceee, rtmp1, rtmp2);
5563 reg_cache.gen_load(2, buf);
5564 md5_FF(reg_cache, a, b, c, d, 4, S11, 0xf57c0faf, rtmp1, rtmp2);
5565 md5_FF(reg_cache, d, a, b, c, 5, S12, 0x4787c62a, rtmp1, rtmp2);
5566 reg_cache.gen_load(3, buf);
5567 md5_FF(reg_cache, c, d, a, b, 6, S13, 0xa8304613, rtmp1, rtmp2);
5568 md5_FF(reg_cache, b, c, d, a, 7, S14, 0xfd469501, rtmp1, rtmp2);
5569 reg_cache.gen_load(4, buf);
5570 md5_FF(reg_cache, a, b, c, d, 8, S11, 0x698098d8, rtmp1, rtmp2);
5571 md5_FF(reg_cache, d, a, b, c, 9, S12, 0x8b44f7af, rtmp1, rtmp2);
5572 reg_cache.gen_load(5, buf);
5573 md5_FF(reg_cache, c, d, a, b, 10, S13, 0xffff5bb1, rtmp1, rtmp2);
5574 md5_FF(reg_cache, b, c, d, a, 11, S14, 0x895cd7be, rtmp1, rtmp2);
5575 reg_cache.gen_load(6, buf);
5576 md5_FF(reg_cache, a, b, c, d, 12, S11, 0x6b901122, rtmp1, rtmp2);
5577 md5_FF(reg_cache, d, a, b, c, 13, S12, 0xfd987193, rtmp1, rtmp2);
5578 reg_cache.gen_load(7, buf);
5579 md5_FF(reg_cache, c, d, a, b, 14, S13, 0xa679438e, rtmp1, rtmp2);
5580 md5_FF(reg_cache, b, c, d, a, 15, S14, 0x49b40821, rtmp1, rtmp2);
5581
5582 // Round 2
5583 md5_GG(reg_cache, a, b, c, d, 1, S21, 0xf61e2562, rtmp1, rtmp2);
5584 md5_GG(reg_cache, d, a, b, c, 6, S22, 0xc040b340, rtmp1, rtmp2);
5585 md5_GG(reg_cache, c, d, a, b, 11, S23, 0x265e5a51, rtmp1, rtmp2);
5586 md5_GG(reg_cache, b, c, d, a, 0, S24, 0xe9b6c7aa, rtmp1, rtmp2);
5587 md5_GG(reg_cache, a, b, c, d, 5, S21, 0xd62f105d, rtmp1, rtmp2);
5588 md5_GG(reg_cache, d, a, b, c, 10, S22, 0x02441453, rtmp1, rtmp2);
5589 md5_GG(reg_cache, c, d, a, b, 15, S23, 0xd8a1e681, rtmp1, rtmp2);
5590 md5_GG(reg_cache, b, c, d, a, 4, S24, 0xe7d3fbc8, rtmp1, rtmp2);
5591 md5_GG(reg_cache, a, b, c, d, 9, S21, 0x21e1cde6, rtmp1, rtmp2);
5592 md5_GG(reg_cache, d, a, b, c, 14, S22, 0xc33707d6, rtmp1, rtmp2);
5593 md5_GG(reg_cache, c, d, a, b, 3, S23, 0xf4d50d87, rtmp1, rtmp2);
5594 md5_GG(reg_cache, b, c, d, a, 8, S24, 0x455a14ed, rtmp1, rtmp2);
5595 md5_GG(reg_cache, a, b, c, d, 13, S21, 0xa9e3e905, rtmp1, rtmp2);
5596 md5_GG(reg_cache, d, a, b, c, 2, S22, 0xfcefa3f8, rtmp1, rtmp2);
5597 md5_GG(reg_cache, c, d, a, b, 7, S23, 0x676f02d9, rtmp1, rtmp2);
5598 md5_GG(reg_cache, b, c, d, a, 12, S24, 0x8d2a4c8a, rtmp1, rtmp2);
5599
5600 // Round 3
5601 md5_HH(reg_cache, a, b, c, d, 5, S31, 0xfffa3942, rtmp1, rtmp2);
5602 md5_HH(reg_cache, d, a, b, c, 8, S32, 0x8771f681, rtmp1, rtmp2);
5603 md5_HH(reg_cache, c, d, a, b, 11, S33, 0x6d9d6122, rtmp1, rtmp2);
5604 md5_HH(reg_cache, b, c, d, a, 14, S34, 0xfde5380c, rtmp1, rtmp2);
5605 md5_HH(reg_cache, a, b, c, d, 1, S31, 0xa4beea44, rtmp1, rtmp2);
5606 md5_HH(reg_cache, d, a, b, c, 4, S32, 0x4bdecfa9, rtmp1, rtmp2);
5607 md5_HH(reg_cache, c, d, a, b, 7, S33, 0xf6bb4b60, rtmp1, rtmp2);
5608 md5_HH(reg_cache, b, c, d, a, 10, S34, 0xbebfbc70, rtmp1, rtmp2);
5609 md5_HH(reg_cache, a, b, c, d, 13, S31, 0x289b7ec6, rtmp1, rtmp2);
5610 md5_HH(reg_cache, d, a, b, c, 0, S32, 0xeaa127fa, rtmp1, rtmp2);
5611 md5_HH(reg_cache, c, d, a, b, 3, S33, 0xd4ef3085, rtmp1, rtmp2);
5612 md5_HH(reg_cache, b, c, d, a, 6, S34, 0x04881d05, rtmp1, rtmp2);
5613 md5_HH(reg_cache, a, b, c, d, 9, S31, 0xd9d4d039, rtmp1, rtmp2);
5614 md5_HH(reg_cache, d, a, b, c, 12, S32, 0xe6db99e5, rtmp1, rtmp2);
5615 md5_HH(reg_cache, c, d, a, b, 15, S33, 0x1fa27cf8, rtmp1, rtmp2);
5616 md5_HH(reg_cache, b, c, d, a, 2, S34, 0xc4ac5665, rtmp1, rtmp2);
5617
5618 // Round 4
5619 md5_II(reg_cache, a, b, c, d, 0, S41, 0xf4292244, rtmp1, rtmp2);
5620 md5_II(reg_cache, d, a, b, c, 7, S42, 0x432aff97, rtmp1, rtmp2);
5621 md5_II(reg_cache, c, d, a, b, 14, S43, 0xab9423a7, rtmp1, rtmp2);
5622 md5_II(reg_cache, b, c, d, a, 5, S44, 0xfc93a039, rtmp1, rtmp2);
5623 md5_II(reg_cache, a, b, c, d, 12, S41, 0x655b59c3, rtmp1, rtmp2);
5624 md5_II(reg_cache, d, a, b, c, 3, S42, 0x8f0ccc92, rtmp1, rtmp2);
5625 md5_II(reg_cache, c, d, a, b, 10, S43, 0xffeff47d, rtmp1, rtmp2);
5626 md5_II(reg_cache, b, c, d, a, 1, S44, 0x85845dd1, rtmp1, rtmp2);
5627 md5_II(reg_cache, a, b, c, d, 8, S41, 0x6fa87e4f, rtmp1, rtmp2);
5628 md5_II(reg_cache, d, a, b, c, 15, S42, 0xfe2ce6e0, rtmp1, rtmp2);
5629 md5_II(reg_cache, c, d, a, b, 6, S43, 0xa3014314, rtmp1, rtmp2);
5630 md5_II(reg_cache, b, c, d, a, 13, S44, 0x4e0811a1, rtmp1, rtmp2);
5631 md5_II(reg_cache, a, b, c, d, 4, S41, 0xf7537e82, rtmp1, rtmp2);
5632 md5_II(reg_cache, d, a, b, c, 11, S42, 0xbd3af235, rtmp1, rtmp2);
5633 md5_II(reg_cache, c, d, a, b, 2, S43, 0x2ad7d2bb, rtmp1, rtmp2);
5634 md5_II(reg_cache, b, c, d, a, 9, S44, 0xeb86d391, rtmp1, rtmp2);
5635
5636 __ addw(state0, state0, a);
5637 __ addw(state1, state1, b);
5638 __ addw(state2, state2, c);
5639 __ addw(state3, state3, d);
5640
5641 if (multi_block) {
5642 __ addi(buf, buf, 64);
5643 __ addi(ofs, ofs, 64);
5644 // if (ofs <= limit) goto m5_loop
5645 __ bge(limit, ofs, md5_loop);
5646 __ mv(c_rarg0, ofs); // return ofs
5647 }
5648
5649 // to minimize the number of memory operations:
5650 // write back the 4 state 4-byte values in pairs, with a single sd
5651 __ mv(t0, mask32);
5652 __ andr(state0, state0, t0);
5653 __ slli(state1, state1, 32);
5654 __ orr(state0, state0, state1);
5655 __ sd(state0, Address(state));
5656 __ andr(state2, state2, t0);
5657 __ slli(state3, state3, 32);
5658 __ orr(state2, state2, state3);
5659 __ sd(state2, Address(state, 8));
5660
5661 __ pop_reg(saved_regs, sp);
5662 __ ret();
5663
5664 return (address) start;
5665 }
5666
5667 /**
5668 * Perform the quarter round calculations on values contained within four vector registers.
5669 *
5670 * @param aVec the SIMD register containing only the "a" values
5671 * @param bVec the SIMD register containing only the "b" values
5672 * @param cVec the SIMD register containing only the "c" values
5673 * @param dVec the SIMD register containing only the "d" values
5674 * @param tmp_vr temporary vector register holds intermedia values.
5675 */
5676 void chacha20_quarter_round(VectorRegister aVec, VectorRegister bVec,
5677 VectorRegister cVec, VectorRegister dVec, VectorRegister tmp_vr) {
5678 // a += b, d ^= a, d <<<= 16
5679 __ vadd_vv(aVec, aVec, bVec);
5680 __ vxor_vv(dVec, dVec, aVec);
5681 __ vrole32_vi(dVec, 16, tmp_vr);
5682
5683 // c += d, b ^= c, b <<<= 12
5684 __ vadd_vv(cVec, cVec, dVec);
5685 __ vxor_vv(bVec, bVec, cVec);
5686 __ vrole32_vi(bVec, 12, tmp_vr);
5687
5688 // a += b, d ^= a, d <<<= 8
5689 __ vadd_vv(aVec, aVec, bVec);
5690 __ vxor_vv(dVec, dVec, aVec);
5691 __ vrole32_vi(dVec, 8, tmp_vr);
5692
5693 // c += d, b ^= c, b <<<= 7
5694 __ vadd_vv(cVec, cVec, dVec);
5695 __ vxor_vv(bVec, bVec, cVec);
5696 __ vrole32_vi(bVec, 7, tmp_vr);
5697 }
5698
5699 /**
5700 * int com.sun.crypto.provider.ChaCha20Cipher.implChaCha20Block(int[] initState, byte[] result)
5701 *
5702 * Input arguments:
5703 * c_rarg0 - state, the starting state
5704 * c_rarg1 - key_stream, the array that will hold the result of the ChaCha20 block function
5705 *
5706 * Implementation Note:
5707 * Parallelization is achieved by loading individual state elements into vectors for N blocks.
5708 * N depends on single vector register length.
5709 */
5710 address generate_chacha20Block() {
5711 Label L_Rounds;
5712
5713 __ align(CodeEntryAlignment);
5714 StubId stub_id = StubId::stubgen_chacha20Block_id;
5715 StubCodeMark mark(this, stub_id);
5716 address start = __ pc();
5717 __ enter();
5718
5719 const int states_len = 16;
5720 const int step = 4;
5721 const Register state = c_rarg0;
5722 const Register key_stream = c_rarg1;
5723 const Register tmp_addr = t0;
5724 const Register length = t1;
5725
5726 // Organize vector registers in an array that facilitates
5727 // putting repetitive opcodes into loop structures below.
5728 const VectorRegister work_vrs[16] = {
5729 v0, v1, v2, v3, v4, v5, v6, v7,
5730 v8, v9, v10, v11, v12, v13, v14, v15
5731 };
5732 const VectorRegister tmp_vr = v16;
5733 const VectorRegister counter_vr = v17;
5734
5735 {
5736 // Put 16 here, as com.sun.crypto.providerChaCha20Cipher.KS_MAX_LEN is 1024
5737 // in java level.
5738 __ vsetivli(length, 16, Assembler::e32, Assembler::m1);
5739 }
5740
5741 // Load from source state.
5742 // Every element in source state is duplicated to all elements in the corresponding vector.
5743 __ mv(tmp_addr, state);
5744 for (int i = 0; i < states_len; i += 1) {
5745 __ vlse32_v(work_vrs[i], tmp_addr, zr);
5746 __ addi(tmp_addr, tmp_addr, step);
5747 }
5748 // Adjust counter for every individual block.
5749 __ vid_v(counter_vr);
5750 __ vadd_vv(work_vrs[12], work_vrs[12], counter_vr);
5751
5752 // Perform 10 iterations of the 8 quarter round set
5753 {
5754 const Register loop = t2; // share t2 with other non-overlapping usages.
5755 __ mv(loop, 10);
5756 __ BIND(L_Rounds);
5757
5758 chacha20_quarter_round(work_vrs[0], work_vrs[4], work_vrs[8], work_vrs[12], tmp_vr);
5759 chacha20_quarter_round(work_vrs[1], work_vrs[5], work_vrs[9], work_vrs[13], tmp_vr);
5760 chacha20_quarter_round(work_vrs[2], work_vrs[6], work_vrs[10], work_vrs[14], tmp_vr);
5761 chacha20_quarter_round(work_vrs[3], work_vrs[7], work_vrs[11], work_vrs[15], tmp_vr);
5762
5763 chacha20_quarter_round(work_vrs[0], work_vrs[5], work_vrs[10], work_vrs[15], tmp_vr);
5764 chacha20_quarter_round(work_vrs[1], work_vrs[6], work_vrs[11], work_vrs[12], tmp_vr);
5765 chacha20_quarter_round(work_vrs[2], work_vrs[7], work_vrs[8], work_vrs[13], tmp_vr);
5766 chacha20_quarter_round(work_vrs[3], work_vrs[4], work_vrs[9], work_vrs[14], tmp_vr);
5767
5768 __ subi(loop, loop, 1);
5769 __ bnez(loop, L_Rounds);
5770 }
5771
5772 // Add the original state into the end working state.
5773 // We do this by first duplicating every element in source state array to the corresponding
5774 // vector, then adding it to the post-loop working state.
5775 __ mv(tmp_addr, state);
5776 for (int i = 0; i < states_len; i += 1) {
5777 __ vlse32_v(tmp_vr, tmp_addr, zr);
5778 __ addi(tmp_addr, tmp_addr, step);
5779 __ vadd_vv(work_vrs[i], work_vrs[i], tmp_vr);
5780 }
5781 // Add the counter overlay onto work_vrs[12] at the end.
5782 __ vadd_vv(work_vrs[12], work_vrs[12], counter_vr);
5783
5784 // Store result to key stream.
5785 {
5786 const Register stride = t2; // share t2 with other non-overlapping usages.
5787 // Every block occupies 64 bytes, so we use 64 as stride of the vector store.
5788 __ mv(stride, 64);
5789 for (int i = 0; i < states_len; i += 1) {
5790 __ vsse32_v(work_vrs[i], key_stream, stride);
5791 __ addi(key_stream, key_stream, step);
5792 }
5793 }
5794
5795 // Return length of output key_stream
5796 __ slli(c_rarg0, length, 6);
5797
5798 __ leave();
5799 __ ret();
5800
5801 return (address) start;
5802 }
5803
5804
5805 // ------------------------ SHA-1 intrinsic ------------------------
5806
5807 // K't =
5808 // 5a827999, 0 <= t <= 19
5809 // 6ed9eba1, 20 <= t <= 39
5810 // 8f1bbcdc, 40 <= t <= 59
5811 // ca62c1d6, 60 <= t <= 79
5812 void sha1_prepare_k(Register cur_k, int round) {
5813 assert(round >= 0 && round < 80, "must be");
5814
5815 static const int64_t ks[] = {0x5a827999, 0x6ed9eba1, 0x8f1bbcdc, 0xca62c1d6};
5816 if ((round % 20) == 0) {
5817 __ mv(cur_k, ks[round/20]);
5818 }
5819 }
5820
5821 // W't =
5822 // M't, 0 <= t <= 15
5823 // ROTL'1(W't-3 ^ W't-8 ^ W't-14 ^ W't-16), 16 <= t <= 79
5824 void sha1_prepare_w(Register cur_w, Register ws[], Register buf, int round) {
5825 assert(round >= 0 && round < 80, "must be");
5826
5827 if (round < 16) {
5828 // in the first 16 rounds, in ws[], every register contains 2 W't, e.g.
5829 // in ws[0], high part contains W't-0, low part contains W't-1,
5830 // in ws[1], high part contains W't-2, low part contains W't-3,
5831 // ...
5832 // in ws[7], high part contains W't-14, low part contains W't-15.
5833
5834 if ((round % 2) == 0) {
5835 __ ld(ws[round/2], Address(buf, (round/2) * 8));
5836 // reverse bytes, as SHA-1 is defined in big-endian.
5837 __ revb(ws[round/2], ws[round/2]);
5838 __ srli(cur_w, ws[round/2], 32);
5839 } else {
5840 __ mv(cur_w, ws[round/2]);
5841 }
5842
5843 return;
5844 }
5845
5846 if ((round % 2) == 0) {
5847 int idx = 16;
5848 // W't = ROTL'1(W't-3 ^ W't-8 ^ W't-14 ^ W't-16), 16 <= t <= 79
5849 __ srli(t1, ws[(idx-8)/2], 32);
5850 __ xorr(t0, ws[(idx-3)/2], t1);
5851
5852 __ srli(t1, ws[(idx-14)/2], 32);
5853 __ srli(cur_w, ws[(idx-16)/2], 32);
5854 __ xorr(cur_w, cur_w, t1);
5855
5856 __ xorr(cur_w, cur_w, t0);
5857 __ rolw(cur_w, cur_w, 1, t0);
5858
5859 // copy the cur_w value to ws[8].
5860 // now, valid w't values are at:
5861 // w0: ws[0]'s lower 32 bits
5862 // w1 ~ w14: ws[1] ~ ws[7]
5863 // w15: ws[8]'s higher 32 bits
5864 __ slli(ws[idx/2], cur_w, 32);
5865
5866 return;
5867 }
5868
5869 int idx = 17;
5870 // W't = ROTL'1(W't-3 ^ W't-8 ^ W't-14 ^ W't-16), 16 <= t <= 79
5871 __ srli(t1, ws[(idx-3)/2], 32);
5872 __ xorr(t0, t1, ws[(idx-8)/2]);
5873
5874 __ xorr(cur_w, ws[(idx-16)/2], ws[(idx-14)/2]);
5875
5876 __ xorr(cur_w, cur_w, t0);
5877 __ rolw(cur_w, cur_w, 1, t0);
5878
5879 // copy the cur_w value to ws[8]
5880 __ zext(cur_w, cur_w, 32);
5881 __ orr(ws[idx/2], ws[idx/2], cur_w);
5882
5883 // shift the w't registers, so they start from ws[0] again.
5884 // now, valid w't values are at:
5885 // w0 ~ w15: ws[0] ~ ws[7]
5886 Register ws_0 = ws[0];
5887 for (int i = 0; i < 16/2; i++) {
5888 ws[i] = ws[i+1];
5889 }
5890 ws[8] = ws_0;
5891 }
5892
5893 // f't(x, y, z) =
5894 // Ch(x, y, z) = (x & y) ^ (~x & z) , 0 <= t <= 19
5895 // Parity(x, y, z) = x ^ y ^ z , 20 <= t <= 39
5896 // Maj(x, y, z) = (x & y) ^ (x & z) ^ (y & z) , 40 <= t <= 59
5897 // Parity(x, y, z) = x ^ y ^ z , 60 <= t <= 79
5898 void sha1_f(Register dst, Register x, Register y, Register z, int round) {
5899 assert(round >= 0 && round < 80, "must be");
5900 assert_different_registers(dst, x, y, z, t0, t1);
5901
5902 if (round < 20) {
5903 // (x & y) ^ (~x & z)
5904 __ andr(t0, x, y);
5905 __ andn(dst, z, x);
5906 __ xorr(dst, dst, t0);
5907 } else if (round >= 40 && round < 60) {
5908 // (x & y) ^ (x & z) ^ (y & z)
5909 __ andr(t0, x, y);
5910 __ andr(t1, x, z);
5911 __ andr(dst, y, z);
5912 __ xorr(dst, dst, t0);
5913 __ xorr(dst, dst, t1);
5914 } else {
5915 // x ^ y ^ z
5916 __ xorr(dst, x, y);
5917 __ xorr(dst, dst, z);
5918 }
5919 }
5920
5921 // T = ROTL'5(a) + f't(b, c, d) + e + K't + W't
5922 // e = d
5923 // d = c
5924 // c = ROTL'30(b)
5925 // b = a
5926 // a = T
5927 void sha1_process_round(Register a, Register b, Register c, Register d, Register e,
5928 Register cur_k, Register cur_w, Register tmp, int round) {
5929 assert(round >= 0 && round < 80, "must be");
5930 assert_different_registers(a, b, c, d, e, cur_w, cur_k, tmp, t0);
5931
5932 // T = ROTL'5(a) + f't(b, c, d) + e + K't + W't
5933
5934 // cur_w will be recalculated at the beginning of each round,
5935 // so, we can reuse it as a temp register here.
5936 Register tmp2 = cur_w;
5937
5938 // reuse e as a temporary register, as we will mv new value into it later
5939 Register tmp3 = e;
5940 __ add(tmp2, cur_k, tmp2);
5941 __ add(tmp3, tmp3, tmp2);
5942 __ rolw(tmp2, a, 5, t0);
5943
5944 sha1_f(tmp, b, c, d, round);
5945
5946 __ add(tmp2, tmp2, tmp);
5947 __ add(tmp2, tmp2, tmp3);
5948
5949 // e = d
5950 // d = c
5951 // c = ROTL'30(b)
5952 // b = a
5953 // a = T
5954 __ mv(e, d);
5955 __ mv(d, c);
5956
5957 __ rolw(c, b, 30);
5958 __ mv(b, a);
5959 __ mv(a, tmp2);
5960 }
5961
5962 // H(i)0 = a + H(i-1)0
5963 // H(i)1 = b + H(i-1)1
5964 // H(i)2 = c + H(i-1)2
5965 // H(i)3 = d + H(i-1)3
5966 // H(i)4 = e + H(i-1)4
5967 void sha1_calculate_im_hash(Register a, Register b, Register c, Register d, Register e,
5968 Register prev_ab, Register prev_cd, Register prev_e) {
5969 assert_different_registers(a, b, c, d, e, prev_ab, prev_cd, prev_e);
5970
5971 __ add(a, a, prev_ab);
5972 __ srli(prev_ab, prev_ab, 32);
5973 __ add(b, b, prev_ab);
5974
5975 __ add(c, c, prev_cd);
5976 __ srli(prev_cd, prev_cd, 32);
5977 __ add(d, d, prev_cd);
5978
5979 __ add(e, e, prev_e);
5980 }
5981
5982 void sha1_preserve_prev_abcde(Register a, Register b, Register c, Register d, Register e,
5983 Register prev_ab, Register prev_cd, Register prev_e) {
5984 assert_different_registers(a, b, c, d, e, prev_ab, prev_cd, prev_e, t0);
5985
5986 __ slli(t0, b, 32);
5987 __ zext(prev_ab, a, 32);
5988 __ orr(prev_ab, prev_ab, t0);
5989
5990 __ slli(t0, d, 32);
5991 __ zext(prev_cd, c, 32);
5992 __ orr(prev_cd, prev_cd, t0);
5993
5994 __ mv(prev_e, e);
5995 }
5996
5997 // Intrinsic for:
5998 // void sun.security.provider.SHA.implCompress0(byte[] buf, int ofs)
5999 // void sun.security.provider.DigestBase.implCompressMultiBlock0(byte[] b, int ofs, int limit)
6000 //
6001 // Arguments:
6002 //
6003 // Inputs:
6004 // c_rarg0: byte[] src array + offset
6005 // c_rarg1: int[] SHA.state
6006 // - - - - - - below are only for implCompressMultiBlock0 - - - - - -
6007 // c_rarg2: int offset
6008 // c_rarg3: int limit
6009 //
6010 // Outputs:
6011 // - - - - - - below are only for implCompressMultiBlock0 - - - - - -
6012 // c_rarg0: int offset, when (multi_block == true)
6013 //
6014 address generate_sha1_implCompress(StubId stub_id) {
6015 bool multi_block;
6016 switch (stub_id) {
6017 case StubId::stubgen_sha1_implCompress_id:
6018 multi_block = false;
6019 break;
6020 case StubId::stubgen_sha1_implCompressMB_id:
6021 multi_block = true;
6022 break;
6023 default:
6024 ShouldNotReachHere();
6025 };
6026 __ align(CodeEntryAlignment);
6027 StubCodeMark mark(this, stub_id);
6028
6029 address start = __ pc();
6030 __ enter();
6031
6032 RegSet saved_regs = RegSet::range(x18, x27);
6033 if (multi_block) {
6034 // use x9 as src below.
6035 saved_regs += RegSet::of(x9);
6036 }
6037 __ push_reg(saved_regs, sp);
6038
6039 // c_rarg0 - c_rarg3: x10 - x13
6040 Register buf = c_rarg0;
6041 Register state = c_rarg1;
6042 Register offset = c_rarg2;
6043 Register limit = c_rarg3;
6044 // use src to contain the original start point of the array.
6045 Register src = x9;
6046
6047 if (multi_block) {
6048 __ sub(limit, limit, offset);
6049 __ add(limit, limit, buf);
6050 __ sub(src, buf, offset);
6051 }
6052
6053 // [args-reg]: x14 - x17
6054 // [temp-reg]: x28 - x31
6055 // [saved-reg]: x18 - x27
6056
6057 // h0/1/2/3/4
6058 const Register a = x14, b = x15, c = x16, d = x17, e = x28;
6059 // w0, w1, ... w15
6060 // put two adjecent w's in one register:
6061 // one at high word part, another at low word part
6062 // at different round (even or odd), w't value reside in different items in ws[].
6063 // w0 ~ w15, either reside in
6064 // ws[0] ~ ws[7], where
6065 // w0 at higher 32 bits of ws[0],
6066 // w1 at lower 32 bits of ws[0],
6067 // ...
6068 // w14 at higher 32 bits of ws[7],
6069 // w15 at lower 32 bits of ws[7].
6070 // or, reside in
6071 // w0: ws[0]'s lower 32 bits
6072 // w1 ~ w14: ws[1] ~ ws[7]
6073 // w15: ws[8]'s higher 32 bits
6074 Register ws[9] = {x29, x30, x31, x18,
6075 x19, x20, x21, x22,
6076 x23}; // auxiliary register for calculating w's value
6077 // current k't's value
6078 const Register cur_k = x24;
6079 // current w't's value
6080 const Register cur_w = x25;
6081 // values of a, b, c, d, e in the previous round
6082 const Register prev_ab = x26, prev_cd = x27;
6083 const Register prev_e = offset; // reuse offset/c_rarg2
6084
6085 // load 5 words state into a, b, c, d, e.
6086 //
6087 // To minimize the number of memory operations, we apply following
6088 // optimization: read the states (a/b/c/d) of 4-byte values in pairs,
6089 // with a single ld, and split them into 2 registers.
6090 //
6091 // And, as the core algorithm of SHA-1 works on 32-bits words, so
6092 // in the following code, it does not care about the content of
6093 // higher 32-bits in a/b/c/d/e. Based on this observation,
6094 // we can apply further optimization, which is to just ignore the
6095 // higher 32-bits in a/c/e, rather than set the higher
6096 // 32-bits of a/c/e to zero explicitly with extra instructions.
6097 __ ld(a, Address(state, 0));
6098 __ srli(b, a, 32);
6099 __ ld(c, Address(state, 8));
6100 __ srli(d, c, 32);
6101 __ lw(e, Address(state, 16));
6102
6103 Label L_sha1_loop;
6104 if (multi_block) {
6105 __ BIND(L_sha1_loop);
6106 }
6107
6108 sha1_preserve_prev_abcde(a, b, c, d, e, prev_ab, prev_cd, prev_e);
6109
6110 for (int round = 0; round < 80; round++) {
6111 // prepare K't value
6112 sha1_prepare_k(cur_k, round);
6113
6114 // prepare W't value
6115 sha1_prepare_w(cur_w, ws, buf, round);
6116
6117 // one round process
6118 sha1_process_round(a, b, c, d, e, cur_k, cur_w, t2, round);
6119 }
6120
6121 // compute the intermediate hash value
6122 sha1_calculate_im_hash(a, b, c, d, e, prev_ab, prev_cd, prev_e);
6123
6124 if (multi_block) {
6125 int64_t block_bytes = 16 * 4;
6126 __ addi(buf, buf, block_bytes);
6127
6128 __ bge(limit, buf, L_sha1_loop, true);
6129 }
6130
6131 // store back the state.
6132 __ zext(a, a, 32);
6133 __ slli(b, b, 32);
6134 __ orr(a, a, b);
6135 __ sd(a, Address(state, 0));
6136 __ zext(c, c, 32);
6137 __ slli(d, d, 32);
6138 __ orr(c, c, d);
6139 __ sd(c, Address(state, 8));
6140 __ sw(e, Address(state, 16));
6141
6142 // return offset
6143 if (multi_block) {
6144 __ sub(c_rarg0, buf, src);
6145 }
6146
6147 __ pop_reg(saved_regs, sp);
6148
6149 __ leave();
6150 __ ret();
6151
6152 return (address) start;
6153 }
6154
6155 /**
6156 * vector registers:
6157 * input VectorRegister's: intputV1-V3, for m2 they could be v2, v4, v6, for m1 they could be v1, v2, v3
6158 * index VectorRegister's: idxV1-V4, for m2 they could be v8, v10, v12, v14, for m1 they could be v4, v5, v6, v7
6159 * output VectorRegister's: outputV1-V4, for m2 they could be v16, v18, v20, v22, for m1 they could be v8, v9, v10, v11
6160 *
6161 * NOTE: each field will occupy a vector register group
6162 */
6163 void base64_vector_encode_round(Register src, Register dst, Register codec,
6164 Register size, Register stepSrc, Register stepDst,
6165 VectorRegister inputV1, VectorRegister inputV2, VectorRegister inputV3,
6166 VectorRegister idxV1, VectorRegister idxV2, VectorRegister idxV3, VectorRegister idxV4,
6167 VectorRegister outputV1, VectorRegister outputV2, VectorRegister outputV3, VectorRegister outputV4,
6168 Assembler::LMUL lmul) {
6169 // set vector register type/len
6170 __ vsetvli(x0, size, Assembler::e8, lmul);
6171
6172 // segmented load src into v registers: mem(src) => vr(3)
6173 __ vlseg3e8_v(inputV1, src);
6174
6175 // src = src + register_group_len_bytes * 3
6176 __ add(src, src, stepSrc);
6177
6178 // encoding
6179 // 1. compute index into lookup table: vr(3) => vr(4)
6180 __ vsrl_vi(idxV1, inputV1, 2);
6181
6182 __ vsrl_vi(idxV2, inputV2, 2);
6183 __ vsll_vi(inputV1, inputV1, 6);
6184 __ vor_vv(idxV2, idxV2, inputV1);
6185 __ vsrl_vi(idxV2, idxV2, 2);
6186
6187 __ vsrl_vi(idxV3, inputV3, 4);
6188 __ vsll_vi(inputV2, inputV2, 4);
6189 __ vor_vv(idxV3, inputV2, idxV3);
6190 __ vsrl_vi(idxV3, idxV3, 2);
6191
6192 __ vsll_vi(idxV4, inputV3, 2);
6193 __ vsrl_vi(idxV4, idxV4, 2);
6194
6195 // 2. indexed load: vr(4) => vr(4)
6196 __ vluxei8_v(outputV1, codec, idxV1);
6197 __ vluxei8_v(outputV2, codec, idxV2);
6198 __ vluxei8_v(outputV3, codec, idxV3);
6199 __ vluxei8_v(outputV4, codec, idxV4);
6200
6201 // segmented store encoded data in v registers back to dst: vr(4) => mem(dst)
6202 __ vsseg4e8_v(outputV1, dst);
6203
6204 // dst = dst + register_group_len_bytes * 4
6205 __ add(dst, dst, stepDst);
6206 }
6207
6208 /**
6209 * void j.u.Base64.Encoder.encodeBlock(byte[] src, int sp, int sl, byte[] dst, int dp, boolean isURL)
6210 *
6211 * Input arguments:
6212 * c_rarg0 - src, source array
6213 * c_rarg1 - sp, src start offset
6214 * c_rarg2 - sl, src end offset
6215 * c_rarg3 - dst, dest array
6216 * c_rarg4 - dp, dst start offset
6217 * c_rarg5 - isURL, Base64 or URL character set
6218 */
6219 address generate_base64_encodeBlock() {
6220 alignas(64) static const char toBase64[64] = {
6221 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
6222 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z',
6223 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm',
6224 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z',
6225 '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '+', '/'
6226 };
6227
6228 alignas(64) static const char toBase64URL[64] = {
6229 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
6230 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z',
6231 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm',
6232 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z',
6233 '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '-', '_'
6234 };
6235
6236 __ align(CodeEntryAlignment);
6237 StubId stub_id = StubId::stubgen_base64_encodeBlock_id;
6238 StubCodeMark mark(this, stub_id);
6239 address start = __ pc();
6240 __ enter();
6241
6242 Register src = c_rarg0;
6243 Register soff = c_rarg1;
6244 Register send = c_rarg2;
6245 Register dst = c_rarg3;
6246 Register doff = c_rarg4;
6247 Register isURL = c_rarg5;
6248
6249 Register codec = c_rarg6;
6250 Register length = c_rarg7; // total length of src data in bytes
6251
6252 Label ProcessData, Exit;
6253
6254 // length should be multiple of 3
6255 __ sub(length, send, soff);
6256 // real src/dst to process data
6257 __ add(src, src, soff);
6258 __ add(dst, dst, doff);
6259
6260 // load the codec base address
6261 __ la(codec, ExternalAddress((address) toBase64));
6262 __ beqz(isURL, ProcessData);
6263 __ la(codec, ExternalAddress((address) toBase64URL));
6264 __ BIND(ProcessData);
6265
6266 // vector version
6267 if (UseRVV) {
6268 Label ProcessM2, ProcessM1, ProcessScalar;
6269
6270 Register size = soff;
6271 Register stepSrcM1 = send;
6272 Register stepSrcM2 = doff;
6273 Register stepDst = isURL;
6274
6275 __ mv(size, MaxVectorSize * 2);
6276 __ mv(stepSrcM1, MaxVectorSize * 3);
6277 __ slli(stepSrcM2, stepSrcM1, 1);
6278 __ mv(stepDst, MaxVectorSize * 2 * 4);
6279
6280 __ blt(length, stepSrcM2, ProcessM1);
6281
6282 __ BIND(ProcessM2);
6283 base64_vector_encode_round(src, dst, codec,
6284 size, stepSrcM2, stepDst,
6285 v2, v4, v6, // inputs
6286 v8, v10, v12, v14, // indexes
6287 v16, v18, v20, v22, // outputs
6288 Assembler::m2);
6289
6290 __ sub(length, length, stepSrcM2);
6291 __ bge(length, stepSrcM2, ProcessM2);
6292
6293 __ BIND(ProcessM1);
6294 __ blt(length, stepSrcM1, ProcessScalar);
6295
6296 __ srli(size, size, 1);
6297 __ srli(stepDst, stepDst, 1);
6298 base64_vector_encode_round(src, dst, codec,
6299 size, stepSrcM1, stepDst,
6300 v1, v2, v3, // inputs
6301 v4, v5, v6, v7, // indexes
6302 v8, v9, v10, v11, // outputs
6303 Assembler::m1);
6304 __ sub(length, length, stepSrcM1);
6305
6306 __ BIND(ProcessScalar);
6307 }
6308
6309 // scalar version
6310 {
6311 Register byte1 = soff, byte0 = send, byte2 = doff;
6312 Register combined24Bits = isURL;
6313
6314 __ beqz(length, Exit);
6315
6316 Label ScalarLoop;
6317 __ BIND(ScalarLoop);
6318 {
6319 // plain: [byte0[7:0] : byte1[7:0] : byte2[7:0]] =>
6320 // encoded: [byte0[7:2] : byte0[1:0]+byte1[7:4] : byte1[3:0]+byte2[7:6] : byte2[5:0]]
6321
6322 // load 3 bytes src data
6323 __ lbu(byte0, Address(src, 0));
6324 __ lbu(byte1, Address(src, 1));
6325 __ lbu(byte2, Address(src, 2));
6326 __ addi(src, src, 3);
6327
6328 // construct 24 bits from 3 bytes
6329 __ slliw(byte0, byte0, 16);
6330 __ slliw(byte1, byte1, 8);
6331 __ orr(combined24Bits, byte0, byte1);
6332 __ orr(combined24Bits, combined24Bits, byte2);
6333
6334 // get codec index and encode(ie. load from codec by index)
6335 __ slliw(byte0, combined24Bits, 8);
6336 __ srliw(byte0, byte0, 26);
6337 __ add(byte0, codec, byte0);
6338 __ lbu(byte0, byte0);
6339
6340 __ slliw(byte1, combined24Bits, 14);
6341 __ srliw(byte1, byte1, 26);
6342 __ add(byte1, codec, byte1);
6343 __ lbu(byte1, byte1);
6344
6345 __ slliw(byte2, combined24Bits, 20);
6346 __ srliw(byte2, byte2, 26);
6347 __ add(byte2, codec, byte2);
6348 __ lbu(byte2, byte2);
6349
6350 __ andi(combined24Bits, combined24Bits, 0x3f);
6351 __ add(combined24Bits, codec, combined24Bits);
6352 __ lbu(combined24Bits, combined24Bits);
6353
6354 // store 4 bytes encoded data
6355 __ sb(byte0, Address(dst, 0));
6356 __ sb(byte1, Address(dst, 1));
6357 __ sb(byte2, Address(dst, 2));
6358 __ sb(combined24Bits, Address(dst, 3));
6359
6360 __ subi(length, length, 3);
6361 __ addi(dst, dst, 4);
6362 // loop back
6363 __ bnez(length, ScalarLoop);
6364 }
6365 }
6366
6367 __ BIND(Exit);
6368
6369 __ leave();
6370 __ ret();
6371
6372 return (address) start;
6373 }
6374
6375 /**
6376 * vector registers:
6377 * input VectorRegister's: intputV1-V4, for m2 they could be v2, v4, v6, for m1 they could be v2, v4, v6, v8
6378 * index VectorRegister's: idxV1-V3, for m2 they could be v8, v10, v12, v14, for m1 they could be v10, v12, v14, v16
6379 * output VectorRegister's: outputV1-V4, for m2 they could be v16, v18, v20, v22, for m1 they could be v18, v20, v22
6380 *
6381 * NOTE: each field will occupy a single vector register group
6382 */
6383 void base64_vector_decode_round(Register src, Register dst, Register codec,
6384 Register size, Register stepSrc, Register stepDst, Register failedIdx,
6385 VectorRegister inputV1, VectorRegister inputV2, VectorRegister inputV3, VectorRegister inputV4,
6386 VectorRegister idxV1, VectorRegister idxV2, VectorRegister idxV3, VectorRegister idxV4,
6387 VectorRegister outputV1, VectorRegister outputV2, VectorRegister outputV3,
6388 Assembler::LMUL lmul) {
6389 // set vector register type/len
6390 __ vsetvli(x0, size, Assembler::e8, lmul, Assembler::ma, Assembler::ta);
6391
6392 // segmented load src into v registers: mem(src) => vr(4)
6393 __ vlseg4e8_v(inputV1, src);
6394
6395 // src = src + register_group_len_bytes * 4
6396 __ add(src, src, stepSrc);
6397
6398 // decoding
6399 // 1. indexed load: vr(4) => vr(4)
6400 __ vluxei8_v(idxV1, codec, inputV1);
6401 __ vluxei8_v(idxV2, codec, inputV2);
6402 __ vluxei8_v(idxV3, codec, inputV3);
6403 __ vluxei8_v(idxV4, codec, inputV4);
6404
6405 // 2. check wrong data
6406 __ vor_vv(outputV1, idxV1, idxV2);
6407 __ vor_vv(outputV2, idxV3, idxV4);
6408 __ vor_vv(outputV1, outputV1, outputV2);
6409 __ vmseq_vi(v0, outputV1, -1);
6410 __ vfirst_m(failedIdx, v0);
6411 Label NoFailure, FailureAtIdx0;
6412 // valid value can only be -1 when < 0
6413 __ bltz(failedIdx, NoFailure);
6414 // when the first data (at index 0) fails, no need to process data anymore
6415 __ beqz(failedIdx, FailureAtIdx0);
6416 __ vsetvli(x0, failedIdx, Assembler::e8, lmul, Assembler::mu, Assembler::tu);
6417 __ slli(stepDst, failedIdx, 1);
6418 __ add(stepDst, failedIdx, stepDst);
6419 __ BIND(NoFailure);
6420
6421 // 3. compute the decoded data: vr(4) => vr(3)
6422 __ vsll_vi(idxV1, idxV1, 2);
6423 __ vsrl_vi(outputV1, idxV2, 4);
6424 __ vor_vv(outputV1, outputV1, idxV1);
6425
6426 __ vsll_vi(idxV2, idxV2, 4);
6427 __ vsrl_vi(outputV2, idxV3, 2);
6428 __ vor_vv(outputV2, outputV2, idxV2);
6429
6430 __ vsll_vi(idxV3, idxV3, 6);
6431 __ vor_vv(outputV3, idxV4, idxV3);
6432
6433 // segmented store encoded data in v registers back to dst: vr(3) => mem(dst)
6434 __ vsseg3e8_v(outputV1, dst);
6435
6436 // dst = dst + register_group_len_bytes * 3
6437 __ add(dst, dst, stepDst);
6438 __ BIND(FailureAtIdx0);
6439 }
6440
6441 /**
6442 * int j.u.Base64.Decoder.decodeBlock(byte[] src, int sp, int sl, byte[] dst, int dp, boolean isURL, boolean isMIME)
6443 *
6444 * Input arguments:
6445 * c_rarg0 - src, source array
6446 * c_rarg1 - sp, src start offset
6447 * c_rarg2 - sl, src end offset
6448 * c_rarg3 - dst, dest array
6449 * c_rarg4 - dp, dst start offset
6450 * c_rarg5 - isURL, Base64 or URL character set
6451 * c_rarg6 - isMIME, Decoding MIME block
6452 */
6453 address generate_base64_decodeBlock() {
6454
6455 static const uint8_t fromBase64[256] = {
6456 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6457 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6458 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 62u, 255u, 255u, 255u, 63u,
6459 52u, 53u, 54u, 55u, 56u, 57u, 58u, 59u, 60u, 61u, 255u, 255u, 255u, 255u, 255u, 255u,
6460 255u, 0u, 1u, 2u, 3u, 4u, 5u, 6u, 7u, 8u, 9u, 10u, 11u, 12u, 13u, 14u,
6461 15u, 16u, 17u, 18u, 19u, 20u, 21u, 22u, 23u, 24u, 25u, 255u, 255u, 255u, 255u, 255u,
6462 255u, 26u, 27u, 28u, 29u, 30u, 31u, 32u, 33u, 34u, 35u, 36u, 37u, 38u, 39u, 40u,
6463 41u, 42u, 43u, 44u, 45u, 46u, 47u, 48u, 49u, 50u, 51u, 255u, 255u, 255u, 255u, 255u,
6464 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6465 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6466 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6467 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6468 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6469 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6470 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6471 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6472 };
6473
6474 static const uint8_t fromBase64URL[256] = {
6475 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6476 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6477 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 62u, 255u, 255u,
6478 52u, 53u, 54u, 55u, 56u, 57u, 58u, 59u, 60u, 61u, 255u, 255u, 255u, 255u, 255u, 255u,
6479 255u, 0u, 1u, 2u, 3u, 4u, 5u, 6u, 7u, 8u, 9u, 10u, 11u, 12u, 13u, 14u,
6480 15u, 16u, 17u, 18u, 19u, 20u, 21u, 22u, 23u, 24u, 25u, 255u, 255u, 255u, 255u, 63u,
6481 255u, 26u, 27u, 28u, 29u, 30u, 31u, 32u, 33u, 34u, 35u, 36u, 37u, 38u, 39u, 40u,
6482 41u, 42u, 43u, 44u, 45u, 46u, 47u, 48u, 49u, 50u, 51u, 255u, 255u, 255u, 255u, 255u,
6483 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6484 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6485 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6486 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6487 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6488 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6489 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6490 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u, 255u,
6491 };
6492
6493 __ align(CodeEntryAlignment);
6494 StubId stub_id = StubId::stubgen_base64_decodeBlock_id;
6495 StubCodeMark mark(this, stub_id);
6496 address start = __ pc();
6497 __ enter();
6498
6499 Register src = c_rarg0;
6500 Register soff = c_rarg1;
6501 Register send = c_rarg2;
6502 Register dst = c_rarg3;
6503 Register doff = c_rarg4;
6504 Register isURL = c_rarg5;
6505 Register isMIME = c_rarg6;
6506
6507 Register codec = c_rarg7;
6508 Register dstBackup = t6;
6509 Register length = t3; // total length of src data in bytes
6510
6511 Label ProcessData, Exit;
6512 Label ProcessScalar, ScalarLoop;
6513
6514 // passed in length (send - soff) is guaranteed to be > 4,
6515 // and in this intrinsic we only process data of length in multiple of 4,
6516 // it's not guaranteed to be multiple of 4 by java level, so do it explicitly
6517 __ sub(length, send, soff);
6518 __ andi(length, length, -4);
6519 // real src/dst to process data
6520 __ add(src, src, soff);
6521 __ add(dst, dst, doff);
6522 // backup of dst, used to calculate the return value at exit
6523 __ mv(dstBackup, dst);
6524
6525 // load the codec base address
6526 __ la(codec, ExternalAddress((address) fromBase64));
6527 __ beqz(isURL, ProcessData);
6528 __ la(codec, ExternalAddress((address) fromBase64URL));
6529 __ BIND(ProcessData);
6530
6531 // vector version
6532 if (UseRVV) {
6533 // for MIME case, it has a default length limit of 76 which could be
6534 // different(smaller) from (send - soff), so in MIME case, we go through
6535 // the scalar code path directly.
6536 __ bnez(isMIME, ScalarLoop);
6537
6538 Label ProcessM1, ProcessM2;
6539
6540 Register failedIdx = soff;
6541 Register stepSrcM1 = send;
6542 Register stepSrcM2 = doff;
6543 Register stepDst = isURL;
6544 Register size = t4;
6545
6546 __ mv(size, MaxVectorSize * 2);
6547 __ mv(stepSrcM1, MaxVectorSize * 4);
6548 __ slli(stepSrcM2, stepSrcM1, 1);
6549 __ mv(stepDst, MaxVectorSize * 2 * 3);
6550
6551 __ blt(length, stepSrcM2, ProcessM1);
6552
6553
6554 // Assembler::m2
6555 __ BIND(ProcessM2);
6556 base64_vector_decode_round(src, dst, codec,
6557 size, stepSrcM2, stepDst, failedIdx,
6558 v2, v4, v6, v8, // inputs
6559 v10, v12, v14, v16, // indexes
6560 v18, v20, v22, // outputs
6561 Assembler::m2);
6562 __ sub(length, length, stepSrcM2);
6563
6564 // error check
6565 // valid value of failedIdx can only be -1 when < 0
6566 __ bgez(failedIdx, Exit);
6567
6568 __ bge(length, stepSrcM2, ProcessM2);
6569
6570
6571 // Assembler::m1
6572 __ BIND(ProcessM1);
6573 __ blt(length, stepSrcM1, ProcessScalar);
6574
6575 __ srli(size, size, 1);
6576 __ srli(stepDst, stepDst, 1);
6577 base64_vector_decode_round(src, dst, codec,
6578 size, stepSrcM1, stepDst, failedIdx,
6579 v1, v2, v3, v4, // inputs
6580 v5, v6, v7, v8, // indexes
6581 v9, v10, v11, // outputs
6582 Assembler::m1);
6583 __ sub(length, length, stepSrcM1);
6584
6585 // error check
6586 // valid value of failedIdx can only be -1 when < 0
6587 __ bgez(failedIdx, Exit);
6588
6589 __ BIND(ProcessScalar);
6590 __ beqz(length, Exit);
6591 }
6592
6593 // scalar version
6594 {
6595 Register byte0 = soff, byte1 = send, byte2 = doff, byte3 = isURL;
6596 Register combined32Bits = t4;
6597
6598 // encoded: [byte0[5:0] : byte1[5:0] : byte2[5:0]] : byte3[5:0]] =>
6599 // plain: [byte0[5:0]+byte1[5:4] : byte1[3:0]+byte2[5:2] : byte2[1:0]+byte3[5:0]]
6600 __ BIND(ScalarLoop);
6601
6602 // load 4 bytes encoded src data
6603 __ lbu(byte0, Address(src, 0));
6604 __ lbu(byte1, Address(src, 1));
6605 __ lbu(byte2, Address(src, 2));
6606 __ lbu(byte3, Address(src, 3));
6607 __ addi(src, src, 4);
6608
6609 // get codec index and decode (ie. load from codec by index)
6610 __ add(byte0, codec, byte0);
6611 __ add(byte1, codec, byte1);
6612 __ lb(byte0, Address(byte0, 0));
6613 __ lb(byte1, Address(byte1, 0));
6614 __ add(byte2, codec, byte2);
6615 __ add(byte3, codec, byte3);
6616 __ lb(byte2, Address(byte2, 0));
6617 __ lb(byte3, Address(byte3, 0));
6618 __ slliw(byte0, byte0, 18);
6619 __ slliw(byte1, byte1, 12);
6620 __ orr(byte0, byte0, byte1);
6621 __ orr(byte0, byte0, byte3);
6622 __ slliw(byte2, byte2, 6);
6623 // For performance consideration, `combined32Bits` is constructed for 2 purposes at the same time,
6624 // 1. error check below
6625 // 2. decode below
6626 __ orr(combined32Bits, byte0, byte2);
6627
6628 // error check
6629 __ bltz(combined32Bits, Exit);
6630
6631 // store 3 bytes decoded data
6632 __ sraiw(byte0, combined32Bits, 16);
6633 __ sraiw(byte1, combined32Bits, 8);
6634 __ sb(byte0, Address(dst, 0));
6635 __ sb(byte1, Address(dst, 1));
6636 __ sb(combined32Bits, Address(dst, 2));
6637
6638 __ subi(length, length, 4);
6639 __ addi(dst, dst, 3);
6640 // loop back
6641 __ bnez(length, ScalarLoop);
6642 }
6643
6644 __ BIND(Exit);
6645 __ sub(c_rarg0, dst, dstBackup);
6646
6647 __ leave();
6648 __ ret();
6649
6650 return (address) start;
6651 }
6652
6653 void adler32_process_bytes(Register buff, Register s1, Register s2, VectorRegister vtable,
6654 VectorRegister vzero, VectorRegister vbytes, VectorRegister vs1acc, VectorRegister vs2acc,
6655 Register temp0, Register temp1, Register temp2, Register temp3,
6656 VectorRegister vtemp1, VectorRegister vtemp2, int step, Assembler::LMUL lmul) {
6657
6658 assert((lmul == Assembler::m4 && step == 64) ||
6659 (lmul == Assembler::m2 && step == 32) ||
6660 (lmul == Assembler::m1 && step == 16),
6661 "LMUL should be aligned with step: m4 and 64, m2 and 32 or m1 and 16");
6662 // Below is function for calculating Adler32 checksum with 64-, 32- or 16-byte step. LMUL=m4, m2 or m1 is used.
6663 // The results are in v12, v13, ..., v22, v23. Example below is for 64-byte step case.
6664 // We use b1, b2, ..., b64 to denote the 64 bytes loaded in each iteration.
6665 // In non-vectorized code, we update s1 and s2 as:
6666 // s1 <- s1 + b1
6667 // s2 <- s2 + s1
6668 // s1 <- s1 + b2
6669 // s2 <- s2 + b1
6670 // ...
6671 // s1 <- s1 + b64
6672 // s2 <- s2 + s1
6673 // Putting above assignments together, we have:
6674 // s1_new = s1 + b1 + b2 + ... + b64
6675 // s2_new = s2 + (s1 + b1) + (s1 + b1 + b2) + ... + (s1 + b1 + b2 + ... + b64) =
6676 // = s2 + s1 * 64 + (b1 * 64 + b2 * 63 + ... + b64 * 1) =
6677 // = s2 + s1 * 64 + (b1, b2, ... b64) dot (64, 63, ... 1)
6678
6679 __ mv(temp3, step);
6680 // Load data
6681 __ vsetvli(temp0, temp3, Assembler::e8, lmul);
6682 __ vle8_v(vbytes, buff);
6683 __ addi(buff, buff, step);
6684
6685 // Upper bound reduction sum for s1_new:
6686 // 0xFF * 64 = 0x3FC0, so:
6687 // 1. Need to do vector-widening reduction sum
6688 // 2. It is safe to perform sign-extension during vmv.x.s with 16-bits elements
6689 __ vwredsumu_vs(vs1acc, vbytes, vzero);
6690 // Multiplication for s2_new
6691 __ vwmulu_vv(vs2acc, vtable, vbytes);
6692
6693 // s2 = s2 + s1 * log2(step)
6694 __ slli(temp1, s1, exact_log2(step));
6695 __ add(s2, s2, temp1);
6696
6697 // Summing up calculated results for s2_new
6698 if (MaxVectorSize > 16) {
6699 __ vsetvli(temp0, temp3, Assembler::e16, lmul);
6700 } else {
6701 // Half of vector-widening multiplication result is in successor of vs2acc
6702 // group for vlen == 16, in which case we need to double vector register
6703 // group width in order to reduction sum all of them
6704 Assembler::LMUL lmulx2 = (lmul == Assembler::m1) ? Assembler::m2 :
6705 (lmul == Assembler::m2) ? Assembler::m4 : Assembler::m8;
6706 __ vsetvli(temp0, temp3, Assembler::e16, lmulx2);
6707 }
6708 // Upper bound for reduction sum:
6709 // 0xFF * (64 + 63 + ... + 2 + 1) = 0x817E0 max for whole register group, so:
6710 // 1. Need to do vector-widening reduction sum
6711 // 2. It is safe to perform sign-extension during vmv.x.s with 32-bits elements
6712 __ vwredsumu_vs(vtemp1, vs2acc, vzero);
6713
6714 // Extracting results for:
6715 // s1_new
6716 __ vmv_x_s(temp0, vs1acc);
6717 __ add(s1, s1, temp0);
6718 // s2_new
6719 __ vsetvli(temp0, temp3, Assembler::e32, Assembler::m1);
6720 __ vmv_x_s(temp1, vtemp1);
6721 __ add(s2, s2, temp1);
6722 }
6723
6724 /***
6725 * int java.util.zip.Adler32.updateBytes(int adler, byte[] b, int off, int len)
6726 *
6727 * Arguments:
6728 *
6729 * Inputs:
6730 * c_rarg0 - int adler
6731 * c_rarg1 - byte* buff (b + off)
6732 * c_rarg2 - int len
6733 *
6734 * Output:
6735 * c_rarg0 - int adler result
6736 */
6737 address generate_updateBytesAdler32() {
6738 __ align(CodeEntryAlignment);
6739 StubId stub_id = StubId::stubgen_updateBytesAdler32_id;
6740 StubCodeMark mark(this, stub_id);
6741 address start = __ pc();
6742
6743 Label L_nmax, L_nmax_loop, L_nmax_loop_entry, L_by16, L_by16_loop,
6744 L_by16_loop_unroll, L_by1_loop, L_do_mod, L_combine, L_by1;
6745
6746 // Aliases
6747 Register adler = c_rarg0;
6748 Register s1 = c_rarg0;
6749 Register s2 = c_rarg3;
6750 Register buff = c_rarg1;
6751 Register len = c_rarg2;
6752 Register nmax = c_rarg4;
6753 Register base = c_rarg5;
6754 Register count = c_rarg6;
6755 Register temp0 = t3;
6756 Register temp1 = t4;
6757 Register temp2 = t5;
6758 Register temp3 = t6;
6759
6760 VectorRegister vzero = v31;
6761 VectorRegister vbytes = v8; // group: v8, v9, v10, v11
6762 VectorRegister vs1acc = v12; // group: v12, v13, v14, v15
6763 VectorRegister vs2acc = v16; // group: v16, v17, v18, v19, v20, v21, v22, v23
6764 VectorRegister vtable_64 = v24; // group: v24, v25, v26, v27
6765 VectorRegister vtable_32 = v4; // group: v4, v5
6766 VectorRegister vtable_16 = v30;
6767 VectorRegister vtemp1 = v28;
6768 VectorRegister vtemp2 = v29;
6769
6770 // Max number of bytes we can process before having to take the mod
6771 // 0x15B0 is 5552 in decimal, the largest n such that 255n(n+1)/2 + (n+1)(BASE-1) <= 2^32-1
6772 const uint64_t BASE = 0xfff1;
6773 const uint64_t NMAX = 0x15B0;
6774
6775 // Loops steps
6776 int step_64 = 64;
6777 int step_32 = 32;
6778 int step_16 = 16;
6779 int step_1 = 1;
6780
6781 __ enter(); // Required for proper stackwalking of RuntimeStub frame
6782 __ mv(temp1, 64);
6783 __ vsetvli(temp0, temp1, Assembler::e8, Assembler::m4);
6784
6785 // Generating accumulation coefficients for further calculations
6786 // vtable_64:
6787 __ vid_v(vtemp1);
6788 __ vrsub_vx(vtable_64, vtemp1, temp1);
6789 // vtable_64 group now contains { 0x40, 0x3f, 0x3e, ..., 0x3, 0x2, 0x1 }
6790
6791 // vtable_32:
6792 __ mv(temp1, 32);
6793 __ vsetvli(temp0, temp1, Assembler::e8, Assembler::m2);
6794 __ vid_v(vtemp1);
6795 __ vrsub_vx(vtable_32, vtemp1, temp1);
6796 // vtable_32 group now contains { 0x20, 0x1f, 0x1e, ..., 0x3, 0x2, 0x1 }
6797
6798 __ vsetivli(temp0, 16, Assembler::e8, Assembler::m1);
6799 // vtable_16:
6800 __ mv(temp1, 16);
6801 __ vid_v(vtemp1);
6802 __ vrsub_vx(vtable_16, vtemp1, temp1);
6803 // vtable_16 now contains { 0x10, 0xf, 0xe, ..., 0x3, 0x2, 0x1 }
6804
6805 __ vmv_v_i(vzero, 0);
6806
6807 __ mv(base, BASE);
6808 __ mv(nmax, NMAX);
6809
6810 // s1 is initialized to the lower 16 bits of adler
6811 // s2 is initialized to the upper 16 bits of adler
6812 __ srliw(s2, adler, 16); // s2 = ((adler >> 16) & 0xffff)
6813 __ zext(s1, adler, 16); // s1 = (adler & 0xffff)
6814
6815 // The pipelined loop needs at least 16 elements for 1 iteration
6816 // It does check this, but it is more effective to skip to the cleanup loop
6817 __ mv(temp0, step_16);
6818 __ bgeu(len, temp0, L_nmax);
6819 __ beqz(len, L_combine);
6820
6821 // Jumping to L_by1_loop
6822 __ subi(len, len, step_1);
6823 __ j(L_by1_loop);
6824
6825 __ bind(L_nmax);
6826 __ sub(len, len, nmax);
6827 __ subi(count, nmax, 16);
6828 __ bltz(len, L_by16);
6829
6830 // Align L_nmax loop by 64
6831 __ bind(L_nmax_loop_entry);
6832 __ subi(count, count, 32);
6833
6834 __ bind(L_nmax_loop);
6835 adler32_process_bytes(buff, s1, s2, vtable_64, vzero,
6836 vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6837 vtemp1, vtemp2, step_64, Assembler::m4);
6838 __ subi(count, count, step_64);
6839 __ bgtz(count, L_nmax_loop);
6840
6841 // There are three iterations left to do
6842 adler32_process_bytes(buff, s1, s2, vtable_32, vzero,
6843 vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6844 vtemp1, vtemp2, step_32, Assembler::m2);
6845 adler32_process_bytes(buff, s1, s2, vtable_16, vzero,
6846 vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6847 vtemp1, vtemp2, step_16, Assembler::m1);
6848
6849 // s1 = s1 % BASE
6850 __ remuw(s1, s1, base);
6851 // s2 = s2 % BASE
6852 __ remuw(s2, s2, base);
6853
6854 __ sub(len, len, nmax);
6855 __ subi(count, nmax, 16);
6856 __ bgez(len, L_nmax_loop_entry);
6857
6858 __ bind(L_by16);
6859 __ add(len, len, count);
6860 __ bltz(len, L_by1);
6861 // Trying to unroll
6862 __ mv(temp3, step_64);
6863 __ blt(len, temp3, L_by16_loop);
6864
6865 __ bind(L_by16_loop_unroll);
6866 adler32_process_bytes(buff, s1, s2, vtable_64, vzero,
6867 vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6868 vtemp1, vtemp2, step_64, Assembler::m4);
6869 __ subi(len, len, step_64);
6870 // By now the temp3 should still be 64
6871 __ bge(len, temp3, L_by16_loop_unroll);
6872
6873 __ bind(L_by16_loop);
6874 adler32_process_bytes(buff, s1, s2, vtable_16, vzero,
6875 vbytes, vs1acc, vs2acc, temp0, temp1, temp2, temp3,
6876 vtemp1, vtemp2, step_16, Assembler::m1);
6877 __ subi(len, len, step_16);
6878 __ bgez(len, L_by16_loop);
6879
6880 __ bind(L_by1);
6881 __ addi(len, len, 15);
6882 __ bltz(len, L_do_mod);
6883
6884 __ bind(L_by1_loop);
6885 __ lbu(temp0, Address(buff, 0));
6886 __ addi(buff, buff, step_1);
6887 __ add(s1, temp0, s1);
6888 __ add(s2, s2, s1);
6889 __ subi(len, len, step_1);
6890 __ bgez(len, L_by1_loop);
6891
6892 __ bind(L_do_mod);
6893 // s1 = s1 % BASE
6894 __ remuw(s1, s1, base);
6895 // s2 = s2 % BASE
6896 __ remuw(s2, s2, base);
6897
6898 // Combine lower bits and higher bits
6899 // adler = s1 | (s2 << 16)
6900 __ bind(L_combine);
6901 __ slli(s2, s2, 16);
6902 __ orr(s1, s1, s2);
6903
6904 __ leave(); // Required for proper stackwalking of RuntimeStub frame
6905 __ ret();
6906
6907 return start;
6908 }
6909
6910 #endif // COMPILER2
6911
6912 // x10 = input (float16)
6913 // f10 = result (float)
6914 // t1 = temporary register
6915 address generate_float16ToFloat() {
6916 __ align(CodeEntryAlignment);
6917 StubId stub_id = StubId::stubgen_hf2f_id;
6918 StubCodeMark mark(this, stub_id);
6919 address entry = __ pc();
6920 BLOCK_COMMENT("float16ToFloat:");
6921
6922 FloatRegister dst = f10;
6923 Register src = x10;
6924 Label NaN_SLOW;
6925
6926 assert(VM_Version::supports_float16_float_conversion(), "must");
6927
6928 // On riscv, NaN needs a special process as fcvt does not work in that case.
6929 // On riscv, Inf does not need a special process as fcvt can handle it correctly.
6930 // but we consider to get the slow path to process NaN and Inf at the same time,
6931 // as both of them are rare cases, and if we try to get the slow path to handle
6932 // only NaN case it would sacrifise the performance for normal cases,
6933 // i.e. non-NaN and non-Inf cases.
6934
6935 // check whether it's a NaN or +/- Inf.
6936 __ mv(t0, 0x7c00);
6937 __ andr(t1, src, t0);
6938 // jump to stub processing NaN and Inf cases.
6939 __ beq(t0, t1, NaN_SLOW);
6940
6941 // non-NaN or non-Inf cases, just use built-in instructions.
6942 __ fmv_h_x(dst, src);
6943 __ fcvt_s_h(dst, dst);
6944 __ ret();
6945
6946 __ bind(NaN_SLOW);
6947 // following instructions mainly focus on NaN, as riscv does not handle
6948 // NaN well with fcvt, but the code also works for Inf at the same time.
6949
6950 // construct a NaN in 32 bits from the NaN in 16 bits,
6951 // we need the payloads of non-canonical NaNs to be preserved.
6952 __ mv(t1, 0x7f800000);
6953 // sign-bit was already set via sign-extension if necessary.
6954 __ slli(t0, src, 13);
6955 __ orr(t1, t0, t1);
6956 __ fmv_w_x(dst, t1);
6957
6958 __ ret();
6959 return entry;
6960 }
6961
6962 // f10 = input (float)
6963 // x10 = result (float16)
6964 // f11 = temporary float register
6965 // t1 = temporary register
6966 address generate_floatToFloat16() {
6967 __ align(CodeEntryAlignment);
6968 StubId stub_id = StubId::stubgen_f2hf_id;
6969 StubCodeMark mark(this, stub_id);
6970 address entry = __ pc();
6971 BLOCK_COMMENT("floatToFloat16:");
6972
6973 Register dst = x10;
6974 FloatRegister src = f10, ftmp = f11;
6975 Label NaN_SLOW;
6976
6977 assert(VM_Version::supports_float16_float_conversion(), "must");
6978
6979 // On riscv, NaN needs a special process as fcvt does not work in that case.
6980
6981 // check whether it's a NaN.
6982 // replace fclass with feq as performance optimization.
6983 __ feq_s(t0, src, src);
6984 // jump to stub processing NaN cases.
6985 __ beqz(t0, NaN_SLOW);
6986
6987 // non-NaN cases, just use built-in instructions.
6988 __ fcvt_h_s(ftmp, src);
6989 __ fmv_x_h(dst, ftmp);
6990 __ ret();
6991
6992 __ bind(NaN_SLOW);
6993
6994 __ float_to_float16_NaN(dst, src, t0, t1);
6995
6996 __ ret();
6997 return entry;
6998 }
6999
7000 #ifdef COMPILER2
7001
7002 static const int64_t right_2_bits = right_n_bits(2);
7003 static const int64_t right_3_bits = right_n_bits(3);
7004
7005 // In sun.security.util.math.intpoly.IntegerPolynomial1305, integers
7006 // are represented as long[5], with BITS_PER_LIMB = 26.
7007 // Pack five 26-bit limbs into three 64-bit registers.
7008 void poly1305_pack_26(Register dest0, Register dest1, Register dest2, Register src, Register tmp1, Register tmp2) {
7009 assert_different_registers(dest0, dest1, dest2, src, tmp1, tmp2);
7010
7011 // The goal is to have 128-bit value in dest2:dest1:dest0
7012 __ ld(dest0, Address(src, 0)); // 26 bits in dest0
7013
7014 __ ld(tmp1, Address(src, sizeof(jlong)));
7015 __ slli(tmp1, tmp1, 26);
7016 __ add(dest0, dest0, tmp1); // 52 bits in dest0
7017
7018 __ ld(tmp2, Address(src, 2 * sizeof(jlong)));
7019 __ slli(tmp1, tmp2, 52);
7020 __ add(dest0, dest0, tmp1); // dest0 is full
7021
7022 __ srli(dest1, tmp2, 12); // 14-bit in dest1
7023
7024 __ ld(tmp1, Address(src, 3 * sizeof(jlong)));
7025 __ slli(tmp1, tmp1, 14);
7026 __ add(dest1, dest1, tmp1); // 40-bit in dest1
7027
7028 __ ld(tmp1, Address(src, 4 * sizeof(jlong)));
7029 __ slli(tmp2, tmp1, 40);
7030 __ add(dest1, dest1, tmp2); // dest1 is full
7031
7032 if (dest2->is_valid()) {
7033 __ srli(tmp1, tmp1, 24);
7034 __ mv(dest2, tmp1); // 2 bits in dest2
7035 } else {
7036 #ifdef ASSERT
7037 Label OK;
7038 __ srli(tmp1, tmp1, 24);
7039 __ beq(zr, tmp1, OK); // 2 bits
7040 __ stop("high bits of Poly1305 integer should be zero");
7041 __ should_not_reach_here();
7042 __ bind(OK);
7043 #endif
7044 }
7045 }
7046
7047 // As above, but return only a 128-bit integer, packed into two
7048 // 64-bit registers.
7049 void poly1305_pack_26(Register dest0, Register dest1, Register src, Register tmp1, Register tmp2) {
7050 poly1305_pack_26(dest0, dest1, noreg, src, tmp1, tmp2);
7051 }
7052
7053 // U_2:U_1:U_0: += (U_2 >> 2) * 5
7054 void poly1305_reduce(Register U_2, Register U_1, Register U_0, Register tmp1, Register tmp2) {
7055 assert_different_registers(U_2, U_1, U_0, tmp1, tmp2);
7056
7057 // First, U_2:U_1:U_0 += (U_2 >> 2)
7058 __ srli(tmp1, U_2, 2);
7059 __ cad(U_0, U_0, tmp1, tmp2); // Add tmp1 to U_0 with carry output to tmp2
7060 __ andi(U_2, U_2, right_2_bits); // Clear U_2 except for the lowest two bits
7061 __ cad(U_1, U_1, tmp2, tmp2); // Add carry to U_1 with carry output to tmp2
7062 __ add(U_2, U_2, tmp2);
7063
7064 // Second, U_2:U_1:U_0 += (U_2 >> 2) << 2
7065 __ slli(tmp1, tmp1, 2);
7066 __ cad(U_0, U_0, tmp1, tmp2); // Add tmp1 to U_0 with carry output to tmp2
7067 __ cad(U_1, U_1, tmp2, tmp2); // Add carry to U_1 with carry output to tmp2
7068 __ add(U_2, U_2, tmp2);
7069 }
7070
7071 // Poly1305, RFC 7539
7072 // void com.sun.crypto.provider.Poly1305.processMultipleBlocks(byte[] input, int offset, int length, long[] aLimbs, long[] rLimbs)
7073
7074 // Arguments:
7075 // c_rarg0: input_start -- where the input is stored
7076 // c_rarg1: length
7077 // c_rarg2: acc_start -- where the output will be stored
7078 // c_rarg3: r_start -- where the randomly generated 128-bit key is stored
7079
7080 // See https://loup-vaillant.fr/tutorials/poly1305-design for a
7081 // description of the tricks used to simplify and accelerate this
7082 // computation.
7083
7084 address generate_poly1305_processBlocks() {
7085 __ align(CodeEntryAlignment);
7086 StubId stub_id = StubId::stubgen_poly1305_processBlocks_id;
7087 StubCodeMark mark(this, stub_id);
7088 address start = __ pc();
7089 __ enter();
7090 Label here;
7091
7092 RegSet saved_regs = RegSet::range(x18, x21);
7093 RegSetIterator<Register> regs = (RegSet::range(x14, x31) - RegSet::range(x22, x27)).begin();
7094 __ push_reg(saved_regs, sp);
7095
7096 // Arguments
7097 const Register input_start = c_rarg0, length = c_rarg1, acc_start = c_rarg2, r_start = c_rarg3;
7098
7099 // R_n is the 128-bit randomly-generated key, packed into two
7100 // registers. The caller passes this key to us as long[5], with
7101 // BITS_PER_LIMB = 26.
7102 const Register R_0 = *regs, R_1 = *++regs;
7103 poly1305_pack_26(R_0, R_1, r_start, t1, t2);
7104
7105 // RR_n is (R_n >> 2) * 5
7106 const Register RR_0 = *++regs, RR_1 = *++regs;
7107 __ srli(t1, R_0, 2);
7108 __ shadd(RR_0, t1, t1, t2, 2);
7109 __ srli(t1, R_1, 2);
7110 __ shadd(RR_1, t1, t1, t2, 2);
7111
7112 // U_n is the current checksum
7113 const Register U_0 = *++regs, U_1 = *++regs, U_2 = *++regs;
7114 poly1305_pack_26(U_0, U_1, U_2, acc_start, t1, t2);
7115
7116 static constexpr int BLOCK_LENGTH = 16;
7117 Label DONE, LOOP;
7118
7119 __ mv(t1, BLOCK_LENGTH);
7120 __ blt(length, t1, DONE); {
7121 __ bind(LOOP);
7122
7123 // S_n is to be the sum of U_n and the next block of data
7124 const Register S_0 = *++regs, S_1 = *++regs, S_2 = *++regs;
7125 __ ld(S_0, Address(input_start, 0));
7126 __ ld(S_1, Address(input_start, wordSize));
7127
7128 __ cad(S_0, S_0, U_0, t1); // Add U_0 to S_0 with carry output to t1
7129 __ cadc(S_1, S_1, U_1, t1); // Add U_1 with carry to S_1 with carry output to t1
7130 __ add(S_2, U_2, t1);
7131
7132 __ addi(S_2, S_2, 1);
7133
7134 const Register U_0HI = *++regs, U_1HI = *++regs;
7135
7136 // NB: this logic depends on some of the special properties of
7137 // Poly1305 keys. In particular, because we know that the top
7138 // four bits of R_0 and R_1 are zero, we can add together
7139 // partial products without any risk of needing to propagate a
7140 // carry out.
7141 __ wide_mul(U_0, U_0HI, S_0, R_0);
7142 __ wide_madd(U_0, U_0HI, S_1, RR_1, t1, t2);
7143 __ wide_madd(U_0, U_0HI, S_2, RR_0, t1, t2);
7144
7145 __ wide_mul(U_1, U_1HI, S_0, R_1);
7146 __ wide_madd(U_1, U_1HI, S_1, R_0, t1, t2);
7147 __ wide_madd(U_1, U_1HI, S_2, RR_1, t1, t2);
7148
7149 __ andi(U_2, R_0, right_2_bits);
7150 __ mul(U_2, S_2, U_2);
7151
7152 // Partial reduction mod 2**130 - 5
7153 __ cad(U_1, U_1, U_0HI, t1); // Add U_0HI to U_1 with carry output to t1
7154 __ adc(U_2, U_2, U_1HI, t1);
7155 // Sum is now in U_2:U_1:U_0.
7156
7157 // U_2:U_1:U_0: += (U_2 >> 2) * 5
7158 poly1305_reduce(U_2, U_1, U_0, t1, t2);
7159
7160 __ subi(length, length, BLOCK_LENGTH);
7161 __ addi(input_start, input_start, BLOCK_LENGTH);
7162 __ mv(t1, BLOCK_LENGTH);
7163 __ bge(length, t1, LOOP);
7164 }
7165
7166 // Further reduce modulo 2^130 - 5
7167 poly1305_reduce(U_2, U_1, U_0, t1, t2);
7168
7169 // Unpack the sum into five 26-bit limbs and write to memory.
7170 // First 26 bits is the first limb
7171 __ slli(t1, U_0, 38); // Take lowest 26 bits
7172 __ srli(t1, t1, 38);
7173 __ sd(t1, Address(acc_start)); // First 26-bit limb
7174
7175 // 27-52 bits of U_0 is the second limb
7176 __ slli(t1, U_0, 12); // Take next 27-52 bits
7177 __ srli(t1, t1, 38);
7178 __ sd(t1, Address(acc_start, sizeof (jlong))); // Second 26-bit limb
7179
7180 // Getting 53-64 bits of U_0 and 1-14 bits of U_1 in one register
7181 __ srli(t1, U_0, 52);
7182 __ slli(t2, U_1, 50);
7183 __ srli(t2, t2, 38);
7184 __ add(t1, t1, t2);
7185 __ sd(t1, Address(acc_start, 2 * sizeof (jlong))); // Third 26-bit limb
7186
7187 // Storing 15-40 bits of U_1
7188 __ slli(t1, U_1, 24); // Already used up 14 bits
7189 __ srli(t1, t1, 38); // Clear all other bits from t1
7190 __ sd(t1, Address(acc_start, 3 * sizeof (jlong))); // Fourth 26-bit limb
7191
7192 // Storing 41-64 bits of U_1 and first three bits from U_2 in one register
7193 __ srli(t1, U_1, 40);
7194 __ andi(t2, U_2, right_3_bits);
7195 __ slli(t2, t2, 24);
7196 __ add(t1, t1, t2);
7197 __ sd(t1, Address(acc_start, 4 * sizeof (jlong))); // Fifth 26-bit limb
7198
7199 __ bind(DONE);
7200 __ pop_reg(saved_regs, sp);
7201 __ leave(); // Required for proper stackwalking
7202 __ ret();
7203
7204 return start;
7205 }
7206
7207 address generate_arrays_hashcode_powers_of_31() {
7208 assert(UseRVV, "sanity");
7209 const int lmul = 2;
7210 const int stride = MaxVectorSize / sizeof(jint) * lmul;
7211 __ align(CodeEntryAlignment);
7212 StubCodeMark mark(this, "StubRoutines", "arrays_hashcode_powers_of_31");
7213 address start = __ pc();
7214 for (int i = stride; i >= 0; i--) {
7215 jint power_of_31 = 1;
7216 for (int j = i; j > 0; j--) {
7217 power_of_31 = java_multiply(power_of_31, 31);
7218 }
7219 __ emit_int32(power_of_31);
7220 }
7221
7222 return start;
7223 }
7224
7225 #endif // COMPILER2
7226
7227 /**
7228 * Arguments:
7229 *
7230 * Inputs:
7231 * c_rarg0 - int crc
7232 * c_rarg1 - byte* buf
7233 * c_rarg2 - int length
7234 *
7235 * Output:
7236 * c_rarg0 - int crc result
7237 */
7238 address generate_updateBytesCRC32() {
7239 assert(UseCRC32Intrinsics, "what are we doing here?");
7240
7241 __ align(CodeEntryAlignment);
7242 StubId stub_id = StubId::stubgen_updateBytesCRC32_id;
7243 StubCodeMark mark(this, stub_id);
7244
7245 address start = __ pc();
7246
7247 // input parameters
7248 const Register crc = c_rarg0; // crc
7249 const Register buf = c_rarg1; // source java byte array address
7250 const Register len = c_rarg2; // length
7251
7252 BLOCK_COMMENT("Entry:");
7253 __ enter(); // required for proper stackwalking of RuntimeStub frame
7254
7255 __ kernel_crc32(crc, buf, len,
7256 c_rarg3, c_rarg4, c_rarg5, c_rarg6, // tmp's for tables
7257 c_rarg7, t2, t3, t4, t5, t6); // misc tmps
7258
7259 __ leave(); // required for proper stackwalking of RuntimeStub frame
7260 __ ret();
7261
7262 return start;
7263 }
7264
7265 // exception handler for upcall stubs
7266 address generate_upcall_stub_exception_handler() {
7267 StubId stub_id = StubId::stubgen_upcall_stub_exception_handler_id;
7268 StubCodeMark mark(this, stub_id);
7269 address start = __ pc();
7270
7271 // Native caller has no idea how to handle exceptions,
7272 // so we just crash here. Up to callee to catch exceptions.
7273 __ verify_oop(x10); // return a exception oop in a0
7274 __ rt_call(CAST_FROM_FN_PTR(address, UpcallLinker::handle_uncaught_exception));
7275 __ should_not_reach_here();
7276
7277 return start;
7278 }
7279
7280 // load Method* target of MethodHandle
7281 // j_rarg0 = jobject receiver
7282 // xmethod = Method* result
7283 address generate_upcall_stub_load_target() {
7284
7285 StubId stub_id = StubId::stubgen_upcall_stub_load_target_id;
7286 StubCodeMark mark(this, stub_id);
7287 address start = __ pc();
7288
7289 __ resolve_global_jobject(j_rarg0, t0, t1);
7290 // Load target method from receiver
7291 __ load_heap_oop(xmethod, Address(j_rarg0, java_lang_invoke_MethodHandle::form_offset()), t0, t1);
7292 __ load_heap_oop(xmethod, Address(xmethod, java_lang_invoke_LambdaForm::vmentry_offset()), t0, t1);
7293 __ load_heap_oop(xmethod, Address(xmethod, java_lang_invoke_MemberName::method_offset()), t0, t1);
7294 __ access_load_at(T_ADDRESS, IN_HEAP, xmethod,
7295 Address(xmethod, java_lang_invoke_ResolvedMethodName::vmtarget_offset()),
7296 noreg, noreg);
7297 __ sd(xmethod, Address(xthread, JavaThread::callee_target_offset())); // just in case callee is deoptimized
7298
7299 __ ret();
7300
7301 return start;
7302 }
7303
7304 #undef __
7305
7306 // Initialization
7307 void generate_preuniverse_stubs() {
7308 // preuniverse stubs are not needed for riscv
7309 }
7310
7311 void generate_initial_stubs() {
7312 // Generate initial stubs and initializes the entry points
7313
7314 // entry points that exist in all platforms Note: This is code
7315 // that could be shared among different platforms - however the
7316 // benefit seems to be smaller than the disadvantage of having a
7317 // much more complicated generator structure. See also comment in
7318 // stubRoutines.hpp.
7319
7320 StubRoutines::_forward_exception_entry = generate_forward_exception();
7321
7322 if (UnsafeMemoryAccess::_table == nullptr) {
7323 UnsafeMemoryAccess::create_table(8 + 4); // 8 for copyMemory; 4 for setMemory
7324 }
7325
7326 StubRoutines::_call_stub_entry =
7327 generate_call_stub(StubRoutines::_call_stub_return_address);
7328
7329 // is referenced by megamorphic call
7330 StubRoutines::_catch_exception_entry = generate_catch_exception();
7331
7332 if (UseCRC32Intrinsics) {
7333 StubRoutines::_updateBytesCRC32 = generate_updateBytesCRC32();
7334 }
7335
7336 if (vmIntrinsics::is_intrinsic_available(vmIntrinsics::_float16ToFloat) &&
7337 vmIntrinsics::is_intrinsic_available(vmIntrinsics::_floatToFloat16)) {
7338 StubRoutines::_hf2f = generate_float16ToFloat();
7339 StubRoutines::_f2hf = generate_floatToFloat16();
7340 }
7341 }
7342
7343 void generate_continuation_stubs() {
7344 // Continuation stubs:
7345 StubRoutines::_cont_thaw = generate_cont_thaw();
7346 StubRoutines::_cont_returnBarrier = generate_cont_returnBarrier();
7347 StubRoutines::_cont_returnBarrierExc = generate_cont_returnBarrier_exception();
7348 StubRoutines::_cont_preempt_stub = generate_cont_preempt_stub();
7349 }
7350
7351 void generate_final_stubs() {
7352 // support for verify_oop (must happen after universe_init)
7353 if (VerifyOops) {
7354 StubRoutines::_verify_oop_subroutine_entry = generate_verify_oop();
7355 }
7356
7357 // arraycopy stubs used by compilers
7358 generate_arraycopy_stubs();
7359
7360 StubRoutines::_method_entry_barrier = generate_method_entry_barrier();
7361
7362 #ifdef COMPILER2
7363 if (UseSecondarySupersTable) {
7364 StubRoutines::_lookup_secondary_supers_table_slow_path_stub = generate_lookup_secondary_supers_table_slow_path_stub();
7365 if (!InlineSecondarySupersTest) {
7366 generate_lookup_secondary_supers_table_stub();
7367 }
7368 }
7369 #endif // COMPILER2
7370
7371 StubRoutines::_upcall_stub_exception_handler = generate_upcall_stub_exception_handler();
7372 StubRoutines::_upcall_stub_load_target = generate_upcall_stub_load_target();
7373
7374 StubRoutines::riscv::set_completed();
7375 }
7376
7377 void generate_compiler_stubs() {
7378 #ifdef COMPILER2
7379 if (UseMulAddIntrinsic) {
7380 StubRoutines::_mulAdd = generate_mulAdd();
7381 }
7382
7383 if (UseMultiplyToLenIntrinsic) {
7384 StubRoutines::_multiplyToLen = generate_multiplyToLen();
7385 }
7386
7387 if (UseSquareToLenIntrinsic) {
7388 StubRoutines::_squareToLen = generate_squareToLen();
7389 }
7390
7391 if (UseMontgomeryMultiplyIntrinsic) {
7392 StubId stub_id = StubId::stubgen_montgomeryMultiply_id;
7393 StubCodeMark mark(this, stub_id);
7394 MontgomeryMultiplyGenerator g(_masm, /*squaring*/false);
7395 StubRoutines::_montgomeryMultiply = g.generate_multiply();
7396 }
7397
7398 if (UseMontgomerySquareIntrinsic) {
7399 StubId stub_id = StubId::stubgen_montgomerySquare_id;
7400 StubCodeMark mark(this, stub_id);
7401 MontgomeryMultiplyGenerator g(_masm, /*squaring*/true);
7402 StubRoutines::_montgomerySquare = g.generate_square();
7403 }
7404
7405 if (UseAESIntrinsics) {
7406 StubRoutines::_aescrypt_encryptBlock = generate_aescrypt_encryptBlock();
7407 StubRoutines::_aescrypt_decryptBlock = generate_aescrypt_decryptBlock();
7408 StubRoutines::_cipherBlockChaining_encryptAESCrypt = generate_cipherBlockChaining_encryptAESCrypt();
7409 StubRoutines::_cipherBlockChaining_decryptAESCrypt = generate_cipherBlockChaining_decryptAESCrypt();
7410 }
7411
7412 if (UseAESCTRIntrinsics) {
7413 StubRoutines::_counterMode_AESCrypt = generate_counterMode_AESCrypt();
7414 }
7415
7416 if (UseGHASHIntrinsics) {
7417 StubRoutines::_ghash_processBlocks = generate_ghash_processBlocks();
7418 }
7419
7420 if (UseAESCTRIntrinsics && UseGHASHIntrinsics) {
7421 StubRoutines::_galoisCounterMode_AESCrypt = generate_galoisCounterMode_AESCrypt();
7422 }
7423
7424 if (UsePoly1305Intrinsics) {
7425 StubRoutines::_poly1305_processBlocks = generate_poly1305_processBlocks();
7426 }
7427
7428 if (UseRVV) {
7429 StubRoutines::_bigIntegerLeftShiftWorker = generate_bigIntegerLeftShift();
7430 StubRoutines::_bigIntegerRightShiftWorker = generate_bigIntegerRightShift();
7431 }
7432
7433 if (UseVectorizedHashCodeIntrinsic && UseRVV) {
7434 StubRoutines::riscv::_arrays_hashcode_powers_of_31 = generate_arrays_hashcode_powers_of_31();
7435 }
7436
7437 if (UseSHA256Intrinsics) {
7438 Sha2Generator sha2(_masm, this);
7439 StubRoutines::_sha256_implCompress = sha2.generate_sha256_implCompress(StubId::stubgen_sha256_implCompress_id);
7440 StubRoutines::_sha256_implCompressMB = sha2.generate_sha256_implCompress(StubId::stubgen_sha256_implCompressMB_id);
7441 }
7442
7443 if (UseSHA512Intrinsics) {
7444 Sha2Generator sha2(_masm, this);
7445 StubRoutines::_sha512_implCompress = sha2.generate_sha512_implCompress(StubId::stubgen_sha512_implCompress_id);
7446 StubRoutines::_sha512_implCompressMB = sha2.generate_sha512_implCompress(StubId::stubgen_sha512_implCompressMB_id);
7447 }
7448
7449 if (UseMD5Intrinsics) {
7450 StubRoutines::_md5_implCompress = generate_md5_implCompress(StubId::stubgen_md5_implCompress_id);
7451 StubRoutines::_md5_implCompressMB = generate_md5_implCompress(StubId::stubgen_md5_implCompressMB_id);
7452 }
7453
7454 if (UseChaCha20Intrinsics) {
7455 StubRoutines::_chacha20Block = generate_chacha20Block();
7456 }
7457
7458 if (UseSHA1Intrinsics) {
7459 StubRoutines::_sha1_implCompress = generate_sha1_implCompress(StubId::stubgen_sha1_implCompress_id);
7460 StubRoutines::_sha1_implCompressMB = generate_sha1_implCompress(StubId::stubgen_sha1_implCompressMB_id);
7461 }
7462
7463 if (UseBASE64Intrinsics) {
7464 StubRoutines::_base64_encodeBlock = generate_base64_encodeBlock();
7465 StubRoutines::_base64_decodeBlock = generate_base64_decodeBlock();
7466 }
7467
7468 if (UseAdler32Intrinsics) {
7469 StubRoutines::_updateBytesAdler32 = generate_updateBytesAdler32();
7470 }
7471
7472 generate_compare_long_strings();
7473
7474 generate_string_indexof_stubs();
7475
7476 #endif // COMPILER2
7477 }
7478
7479 public:
7480 StubGenerator(CodeBuffer* code, BlobId blob_id, AOTStubData* stub_data) : StubCodeGenerator(code, blob_id, stub_data) {
7481 switch(blob_id) {
7482 case BlobId::stubgen_preuniverse_id:
7483 generate_preuniverse_stubs();
7484 break;
7485 case BlobId::stubgen_initial_id:
7486 generate_initial_stubs();
7487 break;
7488 case BlobId::stubgen_continuation_id:
7489 generate_continuation_stubs();
7490 break;
7491 case BlobId::stubgen_compiler_id:
7492 generate_compiler_stubs();
7493 break;
7494 case BlobId::stubgen_final_id:
7495 generate_final_stubs();
7496 break;
7497 default:
7498 fatal("unexpected blob id: %s", StubInfo::name(blob_id));
7499 break;
7500 };
7501 }
7502 }; // end class declaration
7503
7504 void StubGenerator_generate(CodeBuffer* code, BlobId blob_id, AOTStubData* stub_data) {
7505 StubGenerator g(code, blob_id, stub_data);
7506 }