1 /*
2 * Copyright (c) 2020, 2026, Oracle and/or its affiliates. All rights reserved.
3 * DO NOT ALTER OR REMOVE COPYRIGHT NOTICES OR THIS FILE HEADER.
4 *
5 * This code is free software; you can redistribute it and/or modify it
6 * under the terms of the GNU General Public License version 2 only, as
7 * published by the Free Software Foundation.
8 *
9 * This code is distributed in the hope that it will be useful, but WITHOUT
10 * ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
11 * FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License
12 * version 2 for more details (a copy is included in the LICENSE file that
13 * accompanied this code).
14 *
15 * You should have received a copy of the GNU General Public License version
16 * 2 along with this work; if not, write to the Free Software Foundation,
17 * Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA.
18 *
19 * Please contact Oracle, 500 Oracle Parkway, Redwood Shores, CA 94065 USA
20 * or visit www.oracle.com if you need additional information or have any
21 * questions.
22 *
23 */
24
25 #include "asm/assembler.hpp"
26 #include "asm/assembler.inline.hpp"
27 #include "gc/shared/barrierSet.hpp"
28 #include "gc/shared/barrierSetAssembler.hpp"
29 #include "oops/methodData.hpp"
30 #include "opto/c2_MacroAssembler.hpp"
31 #include "opto/intrinsicnode.hpp"
32 #include "opto/output.hpp"
33 #include "opto/opcodes.hpp"
34 #include "opto/subnode.hpp"
35 #include "runtime/globals.hpp"
36 #include "runtime/objectMonitor.hpp"
37 #include "runtime/objectMonitorTable.hpp"
38 #include "runtime/stubRoutines.hpp"
39 #include "runtime/synchronizer.hpp"
40 #include "utilities/checkedCast.hpp"
41 #include "utilities/globalDefinitions.hpp"
42 #include "utilities/powerOfTwo.hpp"
43 #include "utilities/sizes.hpp"
44
45 #ifdef PRODUCT
46 #define BLOCK_COMMENT(str) /* nothing */
47 #define STOP(error) stop(error)
48 #else
49 #define BLOCK_COMMENT(str) block_comment(str)
50 #define STOP(error) block_comment(error); stop(error)
51 #endif
52
53 // C2 compiled method's prolog code.
54 // Beware! This sp_inc is NOT the same as the one mentioned in MacroAssembler::remove_frame but only the size
55 // of the extension space + the additional copy of the return address. That means, it doesn't contain the
56 // frame size (where the local and sp_inc are) and the saved RBP.
57 void C2_MacroAssembler::verified_entry(Compile* C, int sp_inc) {
58 if (C->clinit_barrier_on_entry()) {
59 assert(VM_Version::supports_fast_class_init_checks(), "sanity");
60 assert(!C->method()->holder()->is_not_initialized(), "initialization should have been started");
61
62 Label L_skip_barrier;
63 Register klass = rscratch1;
64
65 mov_metadata(klass, C->method()->holder()->constant_encoding());
66 clinit_barrier(klass, &L_skip_barrier /*L_fast_path*/);
67
68 jump(RuntimeAddress(SharedRuntime::get_handle_wrong_method_stub())); // slow path
69
70 bind(L_skip_barrier);
71 }
72
73 int framesize = C->output()->frame_size_in_bytes();
74 int bangsize = C->output()->bang_size_in_bytes();
75 bool fp_mode_24b = false;
76 int stack_bang_size = C->output()->need_stack_bang(bangsize) ? bangsize : 0;
77
78 assert(stack_bang_size >= framesize || stack_bang_size <= 0, "stack bang size incorrect");
79
80 assert((framesize & (StackAlignmentInBytes-1)) == 0, "frame size not aligned");
81 // Remove word for return addr
82 framesize -= wordSize;
83 stack_bang_size -= wordSize;
84
85 // Calls to C2R adapters often do not accept exceptional returns.
86 // We require that their callers must bang for them. But be careful, because
87 // some VM calls (such as call site linkage) can use several kilobytes of
88 // stack. But the stack safety zone should account for that.
89 // See bugs 4446381, 4468289, 4497237.
90 if (stack_bang_size > 0) {
91 generate_stack_overflow_check(stack_bang_size);
92
93 // We always push rbp, so that on return to interpreter rbp, will be
94 // restored correctly and we can correct the stack.
95 push(rbp);
96 #ifdef ASSERT
97 if (sp_inc > 0) {
98 movl(Address(rsp, 0), badRegWordVal);
99 movl(Address(rsp, VMRegImpl::stack_slot_size), badRegWordVal);
100 }
101 #endif
102 // Save caller's stack pointer into RBP if the frame pointer is preserved.
103 if (PreserveFramePointer) {
104 mov(rbp, rsp);
105 }
106 // Remove word for ebp
107 framesize -= wordSize;
108
109 // Create frame
110 if (framesize) {
111 subptr(rsp, framesize);
112 }
113 } else {
114 subptr(rsp, framesize);
115
116 // Save RBP register now.
117 framesize -= wordSize;
118 movptr(Address(rsp, framesize), rbp);
119 #ifdef ASSERT
120 if (sp_inc > 0) {
121 movl(Address(rsp, framesize), badRegWordVal);
122 movl(Address(rsp, framesize + VMRegImpl::stack_slot_size), badRegWordVal);
123 }
124 #endif
125 // Save caller's stack pointer into RBP if the frame pointer is preserved.
126 if (PreserveFramePointer) {
127 movptr(rbp, rsp);
128 if (framesize > 0) {
129 addptr(rbp, framesize);
130 }
131 }
132 }
133
134 if (C->needs_stack_repair()) {
135 // Save stack increment just below the saved rbp (also account for fixed framesize and rbp)
136 assert((sp_inc & (StackAlignmentInBytes-1)) == 0, "stack increment not aligned");
137 movptr(Address(rsp, framesize - wordSize), sp_inc + framesize);
138 }
139
140 if (VerifyStackAtCalls) { // Majik cookie to verify stack depth
141 framesize -= wordSize;
142 movptr(Address(rsp, framesize), (int32_t)0xbadb100d);
143 }
144
145 #ifdef ASSERT
146 if (VerifyStackAtCalls) {
147 Label L;
148 push(rax);
149 mov(rax, rsp);
150 andptr(rax, StackAlignmentInBytes-1);
151 cmpptr(rax, StackAlignmentInBytes-wordSize);
152 pop(rax);
153 jcc(Assembler::equal, L);
154 STOP("Stack is not properly aligned!");
155 bind(L);
156 }
157 #endif
158 }
159
160 void C2_MacroAssembler::entry_barrier() {
161 BarrierSetAssembler* bs = BarrierSet::barrier_set()->barrier_set_assembler();
162 // We put the non-hot code of the nmethod entry barrier out-of-line in a stub.
163 Label dummy_slow_path;
164 Label dummy_continuation;
165 Label* slow_path = &dummy_slow_path;
166 Label* continuation = &dummy_continuation;
167 if (!Compile::current()->output()->in_scratch_emit_size()) {
168 // Use real labels from actual stub when not emitting code for the purpose of measuring its size
169 C2EntryBarrierStub* stub = new (Compile::current()->comp_arena()) C2EntryBarrierStub();
170 Compile::current()->output()->add_stub(stub);
171 slow_path = &stub->entry();
172 continuation = &stub->continuation();
173 }
174 bs->nmethod_entry_barrier(this, slow_path, continuation);
175 }
176
177 inline Assembler::AvxVectorLen C2_MacroAssembler::vector_length_encoding(int vlen_in_bytes) {
178 switch (vlen_in_bytes) {
179 case 4: // fall-through
180 case 8: // fall-through
181 case 16: return Assembler::AVX_128bit;
182 case 32: return Assembler::AVX_256bit;
183 case 64: return Assembler::AVX_512bit;
184
185 default: {
186 ShouldNotReachHere();
187 return Assembler::AVX_NoVec;
188 }
189 }
190 }
191
192 // fast_lock and fast_unlock used by C2
193
194 // Because the transitions from emitted code to the runtime
195 // monitorenter/exit helper stubs are so slow it's critical that
196 // we inline both the lock-stack fast path and the inflated fast path.
197 //
198 // See also: cmpFastLock and cmpFastUnlock.
199 //
200 // What follows is a specialized inline transliteration of the code
201 // in enter() and exit(). If we're concerned about I$ bloat another
202 // option would be to emit TrySlowEnter and TrySlowExit methods
203 // at startup-time. These methods would accept arguments as
204 // (rax,=Obj, rbx=Self, rcx=box, rdx=Scratch) and return success-failure
205 // indications in the icc.ZFlag. fast_lock and fast_unlock would simply
206 // marshal the arguments and emit calls to TrySlowEnter and TrySlowExit.
207 // In practice, however, the # of lock sites is bounded and is usually small.
208 // Besides the call overhead, TrySlowEnter and TrySlowExit might suffer
209 // if the processor uses simple bimodal branch predictors keyed by EIP
210 // Since the helper routines would be called from multiple synchronization
211 // sites.
212 //
213 // An even better approach would be write "MonitorEnter()" and "MonitorExit()"
214 // in java - using j.u.c and unsafe - and just bind the lock and unlock sites
215 // to those specialized methods. That'd give us a mostly platform-independent
216 // implementation that the JITs could optimize and inline at their pleasure.
217 // Done correctly, the only time we'd need to cross to native could would be
218 // to park() or unpark() threads. We'd also need a few more unsafe operators
219 // to (a) prevent compiler-JIT reordering of non-volatile accesses, and
220 // (b) explicit barriers or fence operations.
221 //
222 // TODO:
223 //
224 // * Arrange for C2 to pass "Self" into fast_lock and fast_unlock in one of the registers (scr).
225 // This avoids manifesting the Self pointer in the fast_lock and fast_unlock terminals.
226 // Given TLAB allocation, Self is usually manifested in a register, so passing it into
227 // the lock operators would typically be faster than reifying Self.
228 //
229 // * Ideally I'd define the primitives as:
230 // fast_lock (nax Obj, nax box, EAX tmp, nax scr) where box, tmp and scr are KILLED.
231 // fast_unlock (nax Obj, EAX box, nax tmp) where box and tmp are KILLED
232 // Unfortunately ADLC bugs prevent us from expressing the ideal form.
233 // Instead, we're stuck with a rather awkward and brittle register assignments below.
234 // Furthermore the register assignments are overconstrained, possibly resulting in
235 // sub-optimal code near the synchronization site.
236 //
237 // * Eliminate the sp-proximity tests and just use "== Self" tests instead.
238 // Alternately, use a better sp-proximity test.
239 //
240 // * Currently ObjectMonitor._Owner can hold either an sp value or a (THREAD *) value.
241 // Either one is sufficient to uniquely identify a thread.
242 // TODO: eliminate use of sp in _owner and use get_thread(tr) instead.
243 //
244 // * Intrinsify notify() and notifyAll() for the common cases where the
245 // object is locked by the calling thread but the waitlist is empty.
246 // avoid the expensive JNI call to JVM_Notify() and JVM_NotifyAll().
247 //
248 // * use jccb and jmpb instead of jcc and jmp to improve code density.
249 // But beware of excessive branch density on AMD Opterons.
250 //
251 // * Both fast_lock and fast_unlock set the ICC.ZF to indicate success
252 // or failure of the fast path. If the fast path fails then we pass
253 // control to the slow path, typically in C. In fast_lock and
254 // fast_unlock we often branch to DONE_LABEL, just to find that C2
255 // will emit a conditional branch immediately after the node.
256 // So we have branches to branches and lots of ICC.ZF games.
257 // Instead, it might be better to have C2 pass a "FailureLabel"
258 // into fast_lock and fast_unlock. In the case of success, control
259 // will drop through the node. ICC.ZF is undefined at exit.
260 // In the case of failure, the node will branch directly to the
261 // FailureLabel
262
263 // obj: object to lock
264 // box: on-stack box address -- KILLED
265 // rax: tmp -- KILLED
266 // t : tmp -- KILLED
267 void C2_MacroAssembler::fast_lock(Register obj, Register box, Register rax_reg,
268 Register t, Register thread) {
269 assert(rax_reg == rax, "Used for CAS");
270 assert_different_registers(obj, box, rax_reg, t, thread);
271
272 // Handle inflated monitor.
273 Label inflated;
274 // Finish fast lock successfully. ZF value is irrelevant.
275 Label locked;
276 // Finish fast lock unsuccessfully. MUST jump with ZF == 0
277 Label slow_path;
278
279 if (UseObjectMonitorTable) {
280 // Clear cache in case fast locking succeeds or we need to take the slow-path.
281 movptr(Address(box, BasicLock::object_monitor_cache_offset_in_bytes()), 0);
282 }
283
284 if (DiagnoseSyncOnValueBasedClasses != 0) {
285 load_klass(rax_reg, obj, t);
286 testb(Address(rax_reg, Klass::misc_flags_offset()), KlassFlags::_misc_is_value_based_class);
287 jcc(Assembler::notZero, slow_path);
288 }
289
290 const Register mark = t;
291
292 { // Fast Lock
293
294 Label push;
295
296 const Register top = UseObjectMonitorTable ? rax_reg : box;
297
298 // Load the mark.
299 movptr(mark, Address(obj, oopDesc::mark_offset_in_bytes()));
300
301 // Prefetch top.
302 movl(top, Address(thread, JavaThread::lock_stack_top_offset()));
303
304 // Check for monitor (0b10).
305 testptr(mark, markWord::monitor_value);
306 jcc(Assembler::notZero, inflated);
307
308 // Check if lock-stack is full.
309 cmpl(top, LockStack::end_offset() - 1);
310 jcc(Assembler::greater, slow_path);
311
312 // Check if recursive.
313 cmpptr(obj, Address(thread, top, Address::times_1, -oopSize));
314 jccb(Assembler::equal, push);
315
316 // Try to lock. Transition lock bits 0b01 => 0b00
317 movptr(rax_reg, mark);
318 orptr(rax_reg, markWord::unlocked_value);
319 andptr(mark, ~(int32_t)markWord::unlocked_value);
320 lock(); cmpxchgptr(mark, Address(obj, oopDesc::mark_offset_in_bytes()));
321 jcc(Assembler::notEqual, slow_path);
322
323 if (UseObjectMonitorTable) {
324 // Need to reload top, clobbered by CAS.
325 movl(top, Address(thread, JavaThread::lock_stack_top_offset()));
326 }
327 bind(push);
328 // After successful lock, push object on lock-stack.
329 movptr(Address(thread, top), obj);
330 addl(Address(thread, JavaThread::lock_stack_top_offset()), oopSize);
331 jmp(locked);
332 }
333
334 { // Handle inflated monitor.
335 bind(inflated);
336
337 const Register monitor = t;
338 // Offsets into the current thread's object monitor cache (omc).
339 const ByteSize thr_omc_offset = JavaThread::om_cache_offset();
340 const ByteSize omc_monitor_offset = OMCache::monitor_offset();
341 const ByteSize omc_obj_offset = OMCache::obj_offset();
342
343 if (!UseObjectMonitorTable) {
344 assert(mark == monitor, "should be the same here");
345 } else {
346 const Register hash = t;
347 Label monitor_found;
348
349 // Look for the monitor in the current thread's object monitor cache (omc).
350
351 movptr(monitor, Address(thread, thr_omc_offset + omc_monitor_offset));
352 cmpptr(obj, Address(thread, thr_omc_offset + omc_obj_offset));
353 jccb(Assembler::equal, monitor_found);
354
355 // Look for the monitor in the table.
356
357 // Get the hash code.
358 movptr(hash, Address(obj, oopDesc::mark_offset_in_bytes()));
359 shrq(hash, markWord::hash_shift);
360 andq(hash, markWord::hash_mask);
361
362 // Get the table and calculate the bucket's address.
363 lea(rax_reg, ExternalAddress(ObjectMonitorTable::current_table_address()));
364 movptr(rax_reg, Address(rax_reg));
365 andq(hash, Address(rax_reg, ObjectMonitorTable::table_capacity_mask_offset()));
366 movptr(rax_reg, Address(rax_reg, ObjectMonitorTable::table_buckets_offset()));
367
368 // Read the monitor from the bucket.
369 movptr(monitor, Address(rax_reg, hash, Address::times_ptr));
370
371 // Check if the monitor in the bucket is special (empty, tombstone or removed)
372 cmpptr(monitor, ObjectMonitorTable::SpecialPointerValues::below_is_special);
373 jcc(Assembler::below, slow_path);
374
375 // Check if object matches.
376 movptr(rax_reg, Address(monitor, ObjectMonitor::object_offset()));
377 BarrierSetAssembler* bs_asm = BarrierSet::barrier_set()->barrier_set_assembler();
378 bs_asm->try_peek_weak_handle_in_nmethod(this, rax_reg, rax_reg, slow_path);
379 cmpptr(rax_reg, obj);
380 jcc(Assembler::notEqual, slow_path);
381
382 // Store the monitor in the current thread's object monitor cache (omc).
383 movptr(Address(thread, thr_omc_offset + omc_monitor_offset), monitor);
384 movptr(Address(thread, thr_omc_offset + omc_obj_offset), obj);
385
386 bind(monitor_found);
387 }
388 const ByteSize monitor_tag = in_ByteSize(UseObjectMonitorTable ? 0 : checked_cast<int>(markWord::monitor_value));
389 const Address recursions_address(monitor, ObjectMonitor::recursions_offset() - monitor_tag);
390 const Address owner_address(monitor, ObjectMonitor::owner_offset() - monitor_tag);
391
392 Label monitor_locked;
393 // Lock the monitor.
394
395 if (UseObjectMonitorTable) {
396 // Cache the monitor for unlock before trashing box. On failure to acquire
397 // the lock, the slow path will reset the entry accordingly (see CacheSetter).
398 movptr(Address(box, BasicLock::object_monitor_cache_offset_in_bytes()), monitor);
399 }
400
401 // Try to CAS owner (no owner => current thread's _monitor_owner_id).
402 xorptr(rax_reg, rax_reg);
403 movptr(box, Address(thread, JavaThread::monitor_owner_id_offset()));
404 lock(); cmpxchgptr(box, owner_address);
405 jccb(Assembler::equal, monitor_locked);
406
407 // Check if recursive.
408 cmpptr(box, rax_reg);
409 jccb(Assembler::notEqual, slow_path);
410
411 // Recursive.
412 increment(recursions_address);
413
414 bind(monitor_locked);
415 }
416
417 bind(locked);
418 // Set ZF = 1
419 xorl(rax_reg, rax_reg);
420
421 #ifdef ASSERT
422 // Check that locked label is reached with ZF set.
423 Label zf_correct;
424 Label zf_bad_zero;
425 jcc(Assembler::zero, zf_correct);
426 jmp(zf_bad_zero);
427 #endif
428
429 bind(slow_path);
430 #ifdef ASSERT
431 // Check that slow_path label is reached with ZF not set.
432 jcc(Assembler::notZero, zf_correct);
433 stop("Fast Lock ZF != 0");
434 bind(zf_bad_zero);
435 stop("Fast Lock ZF != 1");
436 bind(zf_correct);
437 #endif
438 // C2 uses the value of ZF to determine the continuation.
439 }
440
441 // obj: object to lock
442 // rax: tmp -- KILLED
443 // t : tmp - cannot be obj nor rax -- KILLED
444 //
445 // Some commentary on balanced locking:
446 //
447 // fast_lock and fast_unlock are emitted only for provably balanced lock sites.
448 // Methods that don't have provably balanced locking are forced to run in the
449 // interpreter - such methods won't be compiled to use fast_lock and fast_unlock.
450 // The interpreter provides two properties:
451 // I1: At return-time the interpreter automatically and quietly unlocks any
452 // objects acquired in the current activation (frame). Recall that the
453 // interpreter maintains an on-stack list of locks currently held by
454 // a frame.
455 // I2: If a method attempts to unlock an object that is not held by the
456 // frame the interpreter throws IMSX.
457 //
458 // Lets say A(), which has provably balanced locking, acquires O and then calls B().
459 // B() doesn't have provably balanced locking so it runs in the interpreter.
460 // Control returns to A() and A() unlocks O. By I1 and I2, above, we know that O
461 // is still locked by A().
462 //
463 // The only other source of unbalanced locking would be JNI. The "Java Native Interface
464 // Specification" states that an object locked by JNI's MonitorEnter should not be
465 // unlocked by "normal" java-level locking and vice-versa. The specification doesn't
466 // specify what will occur if a program engages in such mixed-mode locking, however.
467 // Arguably given that the spec legislates the JNI case as undefined our implementation
468 // could reasonably *avoid* checking owner in fast_unlock().
469 // In the interest of performance we elide m->Owner==Self check in unlock.
470 // A perfectly viable alternative is to elide the owner check except when
471 // Xcheck:jni is enabled.
472
473 void C2_MacroAssembler::fast_unlock(Register obj, Register reg_rax, Register t, Register thread) {
474 assert(reg_rax == rax, "Used for CAS");
475 assert_different_registers(obj, reg_rax, t);
476
477 // Handle inflated monitor.
478 Label inflated, inflated_check_lock_stack;
479 // Finish fast unlock successfully. MUST jump with ZF == 1
480 Label unlocked, slow_path;
481
482 const Register mark = t;
483 const Register monitor = t;
484 const Register top = UseObjectMonitorTable ? t : reg_rax;
485 const Register box = reg_rax;
486
487 Label dummy;
488 C2FastUnlockStub* stub = nullptr;
489
490 if (!Compile::current()->output()->in_scratch_emit_size()) {
491 stub = new (Compile::current()->comp_arena()) C2FastUnlockStub(obj, mark, reg_rax, thread);
492 Compile::current()->output()->add_stub(stub);
493 }
494
495 Label& push_and_slow_path = stub == nullptr ? dummy : stub->push_and_slow_path();
496
497 { // Fast Unlock
498
499 // Load top.
500 movl(top, Address(thread, JavaThread::lock_stack_top_offset()));
501
502 if (!UseObjectMonitorTable) {
503 // Prefetch mark.
504 movptr(mark, Address(obj, oopDesc::mark_offset_in_bytes()));
505 }
506
507 // Check if obj is top of lock-stack.
508 cmpptr(obj, Address(thread, top, Address::times_1, -oopSize));
509 // Top of lock stack was not obj. Must be monitor.
510 jcc(Assembler::notEqual, inflated_check_lock_stack);
511
512 // Pop lock-stack.
513 DEBUG_ONLY(movptr(Address(thread, top, Address::times_1, -oopSize), 0);)
514 subl(Address(thread, JavaThread::lock_stack_top_offset()), oopSize);
515
516 // Check if recursive.
517 cmpptr(obj, Address(thread, top, Address::times_1, -2 * oopSize));
518 jcc(Assembler::equal, unlocked);
519
520 // We elide the monitor check, let the CAS fail instead.
521
522 if (UseObjectMonitorTable) {
523 // Load mark.
524 movptr(mark, Address(obj, oopDesc::mark_offset_in_bytes()));
525 }
526
527 // Try to unlock. Transition lock bits 0b00 => 0b01
528 movptr(reg_rax, mark);
529 andptr(reg_rax, ~(int32_t)markWord::lock_mask_in_place);
530 orptr(mark, markWord::unlocked_value);
531 lock(); cmpxchgptr(mark, Address(obj, oopDesc::mark_offset_in_bytes()));
532 jcc(Assembler::notEqual, push_and_slow_path);
533 jmp(unlocked);
534 }
535
536
537 { // Handle inflated monitor.
538 bind(inflated_check_lock_stack);
539 #ifdef ASSERT
540 Label check_done;
541 subl(top, oopSize);
542 cmpl(top, in_bytes(JavaThread::lock_stack_base_offset()));
543 jcc(Assembler::below, check_done);
544 cmpptr(obj, Address(thread, top));
545 jcc(Assembler::notEqual, inflated_check_lock_stack);
546 stop("Fast Unlock lock on stack");
547 bind(check_done);
548 if (UseObjectMonitorTable) {
549 movptr(mark, Address(obj, oopDesc::mark_offset_in_bytes()));
550 }
551 testptr(mark, markWord::monitor_value);
552 jcc(Assembler::notZero, inflated);
553 stop("Fast Unlock not monitor");
554 #endif
555
556 bind(inflated);
557
558 if (!UseObjectMonitorTable) {
559 assert(mark == monitor, "should be the same here");
560 } else {
561 // Uses ObjectMonitorTable. Look for the monitor in our BasicLock on the stack.
562 movptr(monitor, Address(box, BasicLock::object_monitor_cache_offset_in_bytes()));
563 // null check with ZF == 0, no valid pointer below alignof(ObjectMonitor*)
564 cmpptr(monitor, alignof(ObjectMonitor*));
565 jcc(Assembler::below, slow_path);
566 }
567 const ByteSize monitor_tag = in_ByteSize(UseObjectMonitorTable ? 0 : checked_cast<int>(markWord::monitor_value));
568 const Address recursions_address{monitor, ObjectMonitor::recursions_offset() - monitor_tag};
569 const Address succ_address{monitor, ObjectMonitor::succ_offset() - monitor_tag};
570 const Address entry_list_address{monitor, ObjectMonitor::entry_list_offset() - monitor_tag};
571 const Address owner_address{monitor, ObjectMonitor::owner_offset() - monitor_tag};
572
573 Label recursive;
574
575 // Check if recursive.
576 cmpptr(recursions_address, 0);
577 jcc(Assembler::notZero, recursive);
578
579 // Set owner to null.
580 // Release to satisfy the JMM
581 movptr(owner_address, NULL_WORD);
582 // We need a full fence after clearing owner to avoid stranding.
583 // StoreLoad achieves this.
584 membar(StoreLoad);
585
586 // Check if the entry_list is empty.
587 cmpptr(entry_list_address, NULL_WORD);
588 jcc(Assembler::zero, unlocked); // If so we are done.
589
590 // Check if there is a successor.
591 cmpptr(succ_address, NULL_WORD);
592 jcc(Assembler::notZero, unlocked); // If so we are done.
593
594 // Save the monitor pointer in the current thread, so we can try to
595 // reacquire the lock in SharedRuntime::monitor_exit_helper().
596 if (!UseObjectMonitorTable) {
597 andptr(monitor, ~(int32_t)markWord::monitor_value);
598 }
599 movptr(Address(thread, JavaThread::unlocked_inflated_monitor_offset()), monitor);
600
601 orl(t, 1); // Fast Unlock ZF = 0
602 jmpb(slow_path);
603
604 // Recursive unlock.
605 bind(recursive);
606 decrement(recursions_address);
607 }
608
609 bind(unlocked);
610 xorl(t, t); // Fast Unlock ZF = 1
611
612 #ifdef ASSERT
613 // Check that unlocked label is reached with ZF set.
614 Label zf_correct;
615 Label zf_bad_zero;
616 jcc(Assembler::zero, zf_correct);
617 jmp(zf_bad_zero);
618 #endif
619
620 bind(slow_path);
621 if (stub != nullptr) {
622 bind(stub->slow_path_continuation());
623 }
624 #ifdef ASSERT
625 // Check that stub->continuation() label is reached with ZF not set.
626 jcc(Assembler::notZero, zf_correct);
627 stop("Fast Unlock ZF != 0");
628 bind(zf_bad_zero);
629 stop("Fast Unlock ZF != 1");
630 bind(zf_correct);
631 #endif
632 // C2 uses the value of ZF to determine the continuation.
633 }
634
635 static void abort_verify_int_in_range(uint idx, jint val, jint lo, jint hi) {
636 fatal("Invalid CastII, idx: %u, val: %d, lo: %d, hi: %d", idx, val, lo, hi);
637 }
638
639 static void reconstruct_frame_pointer_helper(MacroAssembler* masm, Register dst) {
640 const int framesize = Compile::current()->output()->frame_size_in_bytes();
641 masm->movptr(dst, rsp);
642 if (framesize > 2 * wordSize) {
643 masm->addptr(dst, framesize - 2 * wordSize);
644 }
645 }
646
647 void C2_MacroAssembler::reconstruct_frame_pointer(Register rtmp) {
648 if (PreserveFramePointer) {
649 // frame pointer is valid
650 #ifdef ASSERT
651 // Verify frame pointer value in rbp.
652 reconstruct_frame_pointer_helper(this, rtmp);
653 Label L_success;
654 cmpq(rbp, rtmp);
655 jccb(Assembler::equal, L_success);
656 STOP("frame pointer mismatch");
657 bind(L_success);
658 #endif // ASSERT
659 } else {
660 reconstruct_frame_pointer_helper(this, rbp);
661 }
662 }
663
664 void C2_MacroAssembler::verify_int_in_range(uint idx, const TypeInt* t, Register val) {
665 jint lo = t->_lo;
666 jint hi = t->_hi;
667 assert(lo < hi, "type should not be empty or constant, idx: %u, lo: %d, hi: %d", idx, lo, hi);
668 if (t == TypeInt::INT) {
669 return;
670 }
671
672 BLOCK_COMMENT("CastII {");
673 Label fail;
674 Label succeed;
675
676 if (lo != min_jint) {
677 cmpl(val, lo);
678 jccb(Assembler::less, fail);
679 }
680 if (hi != max_jint) {
681 cmpl(val, hi);
682 jccb(Assembler::greater, fail);
683 }
684 jmpb(succeed);
685
686 bind(fail);
687 movl(c_rarg0, idx);
688 movl(c_rarg1, val);
689 movl(c_rarg2, lo);
690 movl(c_rarg3, hi);
691 reconstruct_frame_pointer(rscratch1);
692 call(RuntimeAddress(CAST_FROM_FN_PTR(address, abort_verify_int_in_range)));
693 hlt();
694 bind(succeed);
695 BLOCK_COMMENT("} // CastII");
696 }
697
698 static void abort_verify_long_in_range(uint idx, jlong val, jlong lo, jlong hi) {
699 fatal("Invalid CastLL, idx: %u, val: " JLONG_FORMAT ", lo: " JLONG_FORMAT ", hi: " JLONG_FORMAT, idx, val, lo, hi);
700 }
701
702 void C2_MacroAssembler::verify_long_in_range(uint idx, const TypeLong* t, Register val, Register tmp) {
703 jlong lo = t->_lo;
704 jlong hi = t->_hi;
705 assert(lo < hi, "type should not be empty or constant, idx: %u, lo: " JLONG_FORMAT ", hi: " JLONG_FORMAT, idx, lo, hi);
706 if (t == TypeLong::LONG) {
707 return;
708 }
709
710 BLOCK_COMMENT("CastLL {");
711 Label fail;
712 Label succeed;
713
714 auto cmp_val = [&](jlong bound) {
715 if (is_simm32(bound)) {
716 cmpq(val, checked_cast<int>(bound));
717 } else {
718 mov64(tmp, bound);
719 cmpq(val, tmp);
720 }
721 };
722
723 if (lo != min_jlong) {
724 cmp_val(lo);
725 jccb(Assembler::less, fail);
726 }
727 if (hi != max_jlong) {
728 cmp_val(hi);
729 jccb(Assembler::greater, fail);
730 }
731 jmpb(succeed);
732
733 bind(fail);
734 movl(c_rarg0, idx);
735 movq(c_rarg1, val);
736 mov64(c_rarg2, lo);
737 mov64(c_rarg3, hi);
738 reconstruct_frame_pointer(rscratch1);
739 call(RuntimeAddress(CAST_FROM_FN_PTR(address, abort_verify_long_in_range)));
740 hlt();
741 bind(succeed);
742 BLOCK_COMMENT("} // CastLL");
743 }
744
745 //-------------------------------------------------------------------------------------------
746 // Generic instructions support for use in .ad files C2 code generation
747
748 void C2_MacroAssembler::vabsnegd(int opcode, XMMRegister dst, XMMRegister src) {
749 if (dst != src) {
750 movdqu(dst, src);
751 }
752 if (opcode == Op_AbsVD) {
753 andpd(dst, ExternalAddress(StubRoutines::x86::vector_double_sign_mask()), noreg);
754 } else {
755 assert((opcode == Op_NegVD),"opcode should be Op_NegD");
756 xorpd(dst, ExternalAddress(StubRoutines::x86::vector_double_sign_flip()), noreg);
757 }
758 }
759
760 void C2_MacroAssembler::vabsnegd(int opcode, XMMRegister dst, XMMRegister src, int vector_len) {
761 if (opcode == Op_AbsVD) {
762 vandpd(dst, src, ExternalAddress(StubRoutines::x86::vector_double_sign_mask()), vector_len, noreg);
763 } else {
764 assert((opcode == Op_NegVD),"opcode should be Op_NegD");
765 vxorpd(dst, src, ExternalAddress(StubRoutines::x86::vector_double_sign_flip()), vector_len, noreg);
766 }
767 }
768
769 void C2_MacroAssembler::vabsnegf(int opcode, XMMRegister dst, XMMRegister src) {
770 if (dst != src) {
771 movdqu(dst, src);
772 }
773 if (opcode == Op_AbsVF) {
774 andps(dst, ExternalAddress(StubRoutines::x86::vector_float_sign_mask()), noreg);
775 } else {
776 assert((opcode == Op_NegVF),"opcode should be Op_NegF");
777 xorps(dst, ExternalAddress(StubRoutines::x86::vector_float_sign_flip()), noreg);
778 }
779 }
780
781 void C2_MacroAssembler::vabsnegf(int opcode, XMMRegister dst, XMMRegister src, int vector_len) {
782 if (opcode == Op_AbsVF) {
783 vandps(dst, src, ExternalAddress(StubRoutines::x86::vector_float_sign_mask()), vector_len, noreg);
784 } else {
785 assert((opcode == Op_NegVF),"opcode should be Op_NegF");
786 vxorps(dst, src, ExternalAddress(StubRoutines::x86::vector_float_sign_flip()), vector_len, noreg);
787 }
788 }
789
790 void C2_MacroAssembler::pminmax(int opcode, BasicType elem_bt, XMMRegister dst, XMMRegister src, XMMRegister tmp) {
791 assert(opcode == Op_MinV || opcode == Op_MaxV, "sanity");
792 assert(tmp == xnoreg || elem_bt == T_LONG, "unused");
793
794 if (opcode == Op_MinV) {
795 if (elem_bt == T_BYTE) {
796 pminsb(dst, src);
797 } else if (elem_bt == T_SHORT) {
798 pminsw(dst, src);
799 } else if (elem_bt == T_INT) {
800 pminsd(dst, src);
801 } else {
802 assert(elem_bt == T_LONG, "required");
803 assert(tmp == xmm0, "required");
804 assert_different_registers(dst, src, tmp);
805 movdqu(xmm0, dst);
806 pcmpgtq(xmm0, src);
807 blendvpd(dst, src); // xmm0 as mask
808 }
809 } else { // opcode == Op_MaxV
810 if (elem_bt == T_BYTE) {
811 pmaxsb(dst, src);
812 } else if (elem_bt == T_SHORT) {
813 pmaxsw(dst, src);
814 } else if (elem_bt == T_INT) {
815 pmaxsd(dst, src);
816 } else {
817 assert(elem_bt == T_LONG, "required");
818 assert(tmp == xmm0, "required");
819 assert_different_registers(dst, src, tmp);
820 movdqu(xmm0, src);
821 pcmpgtq(xmm0, dst);
822 blendvpd(dst, src); // xmm0 as mask
823 }
824 }
825 }
826
827 void C2_MacroAssembler::vpuminmax(int opcode, BasicType elem_bt, XMMRegister dst,
828 XMMRegister src1, Address src2, int vlen_enc) {
829 assert(opcode == Op_UMinV || opcode == Op_UMaxV, "sanity");
830 if (opcode == Op_UMinV) {
831 switch(elem_bt) {
832 case T_BYTE: vpminub(dst, src1, src2, vlen_enc); break;
833 case T_SHORT: vpminuw(dst, src1, src2, vlen_enc); break;
834 case T_INT: vpminud(dst, src1, src2, vlen_enc); break;
835 case T_LONG: evpminuq(dst, k0, src1, src2, false, vlen_enc); break;
836 default: fatal("Unsupported type %s", type2name(elem_bt)); break;
837 }
838 } else {
839 assert(opcode == Op_UMaxV, "required");
840 switch(elem_bt) {
841 case T_BYTE: vpmaxub(dst, src1, src2, vlen_enc); break;
842 case T_SHORT: vpmaxuw(dst, src1, src2, vlen_enc); break;
843 case T_INT: vpmaxud(dst, src1, src2, vlen_enc); break;
844 case T_LONG: evpmaxuq(dst, k0, src1, src2, false, vlen_enc); break;
845 default: fatal("Unsupported type %s", type2name(elem_bt)); break;
846 }
847 }
848 }
849
850 void C2_MacroAssembler::vpuminmaxq(int opcode, XMMRegister dst, XMMRegister src1, XMMRegister src2, XMMRegister xtmp1, XMMRegister xtmp2, int vlen_enc) {
851 // For optimality, leverage a full vector width of 512 bits
852 // for operations over smaller vector sizes on AVX512 targets.
853 if (VM_Version::supports_evex() && !VM_Version::supports_avx512vl()) {
854 if (opcode == Op_UMaxV) {
855 evpmaxuq(dst, k0, src1, src2, false, Assembler::AVX_512bit);
856 } else {
857 assert(opcode == Op_UMinV, "required");
858 evpminuq(dst, k0, src1, src2, false, Assembler::AVX_512bit);
859 }
860 } else {
861 // T1 = -1
862 vpcmpeqq(xtmp1, xtmp1, xtmp1, vlen_enc);
863 // T1 = -1 << 63
864 vpsllq(xtmp1, xtmp1, 63, vlen_enc);
865 // Convert SRC2 to signed value i.e. T2 = T1 + SRC2
866 vpaddq(xtmp2, xtmp1, src2, vlen_enc);
867 // Convert SRC1 to signed value i.e. T1 = T1 + SRC1
868 vpaddq(xtmp1, xtmp1, src1, vlen_enc);
869 // Mask = T2 > T1
870 vpcmpgtq(xtmp1, xtmp2, xtmp1, vlen_enc);
871 if (opcode == Op_UMaxV) {
872 // Res = Mask ? Src2 : Src1
873 vpblendvb(dst, src1, src2, xtmp1, vlen_enc);
874 } else {
875 // Res = Mask ? Src1 : Src2
876 vpblendvb(dst, src2, src1, xtmp1, vlen_enc);
877 }
878 }
879 }
880
881 void C2_MacroAssembler::vpuminmax(int opcode, BasicType elem_bt, XMMRegister dst,
882 XMMRegister src1, XMMRegister src2, int vlen_enc) {
883 assert(opcode == Op_UMinV || opcode == Op_UMaxV, "sanity");
884 if (opcode == Op_UMinV) {
885 switch(elem_bt) {
886 case T_BYTE: vpminub(dst, src1, src2, vlen_enc); break;
887 case T_SHORT: vpminuw(dst, src1, src2, vlen_enc); break;
888 case T_INT: vpminud(dst, src1, src2, vlen_enc); break;
889 case T_LONG: evpminuq(dst, k0, src1, src2, false, vlen_enc); break;
890 default: fatal("Unsupported type %s", type2name(elem_bt)); break;
891 }
892 } else {
893 assert(opcode == Op_UMaxV, "required");
894 switch(elem_bt) {
895 case T_BYTE: vpmaxub(dst, src1, src2, vlen_enc); break;
896 case T_SHORT: vpmaxuw(dst, src1, src2, vlen_enc); break;
897 case T_INT: vpmaxud(dst, src1, src2, vlen_enc); break;
898 case T_LONG: evpmaxuq(dst, k0, src1, src2, false, vlen_enc); break;
899 default: fatal("Unsupported type %s", type2name(elem_bt)); break;
900 }
901 }
902 }
903
904 void C2_MacroAssembler::vpminmax(int opcode, BasicType elem_bt,
905 XMMRegister dst, XMMRegister src1, XMMRegister src2,
906 int vlen_enc) {
907 assert(opcode == Op_MinV || opcode == Op_MaxV, "sanity");
908
909 if (opcode == Op_MinV) {
910 if (elem_bt == T_BYTE) {
911 vpminsb(dst, src1, src2, vlen_enc);
912 } else if (elem_bt == T_SHORT) {
913 vpminsw(dst, src1, src2, vlen_enc);
914 } else if (elem_bt == T_INT) {
915 vpminsd(dst, src1, src2, vlen_enc);
916 } else {
917 assert(elem_bt == T_LONG, "required");
918 if (UseAVX > 2 && (vlen_enc == Assembler::AVX_512bit || VM_Version::supports_avx512vl())) {
919 vpminsq(dst, src1, src2, vlen_enc);
920 } else {
921 assert_different_registers(dst, src1, src2);
922 vpcmpgtq(dst, src1, src2, vlen_enc);
923 vblendvpd(dst, src1, src2, dst, vlen_enc);
924 }
925 }
926 } else { // opcode == Op_MaxV
927 if (elem_bt == T_BYTE) {
928 vpmaxsb(dst, src1, src2, vlen_enc);
929 } else if (elem_bt == T_SHORT) {
930 vpmaxsw(dst, src1, src2, vlen_enc);
931 } else if (elem_bt == T_INT) {
932 vpmaxsd(dst, src1, src2, vlen_enc);
933 } else {
934 assert(elem_bt == T_LONG, "required");
935 if (UseAVX > 2 && (vlen_enc == Assembler::AVX_512bit || VM_Version::supports_avx512vl())) {
936 vpmaxsq(dst, src1, src2, vlen_enc);
937 } else {
938 assert_different_registers(dst, src1, src2);
939 vpcmpgtq(dst, src1, src2, vlen_enc);
940 vblendvpd(dst, src2, src1, dst, vlen_enc);
941 }
942 }
943 }
944 }
945
946 // Float/Double min max
947
948 void C2_MacroAssembler::vminmax_fp(int opcode, BasicType elem_bt,
949 XMMRegister dst, XMMRegister a, XMMRegister b,
950 XMMRegister tmp, XMMRegister atmp, XMMRegister btmp,
951 int vlen_enc) {
952 assert(UseAVX > 0, "required");
953 assert(opcode == Op_MinV || opcode == Op_MinReductionV ||
954 opcode == Op_MaxV || opcode == Op_MaxReductionV, "sanity");
955 assert(elem_bt == T_FLOAT || elem_bt == T_DOUBLE, "sanity");
956 assert_different_registers(a, tmp, atmp, btmp);
957 assert_different_registers(b, tmp, atmp, btmp);
958
959 bool is_min = (opcode == Op_MinV || opcode == Op_MinReductionV);
960 bool is_double_word = is_double_word_type(elem_bt);
961
962 /* Note on 'non-obvious' assembly sequence:
963 *
964 * While there are vminps/vmaxps instructions, there are two important differences between hardware
965 * and Java on how they handle floats:
966 * a. -0.0 and +0.0 are considered equal (vminps/vmaxps will return second parameter when inputs are equal)
967 * b. NaN is not necesarily propagated (vminps/vmaxps will return second parameter when either input is NaN)
968 *
969 * It is still more efficient to use vminps/vmaxps, but with some pre/post-processing:
970 * a. -0.0/+0.0: Bias negative (positive) numbers to second parameter before vminps (vmaxps)
971 * (only useful when signs differ, noop otherwise)
972 * b. NaN: Check if it was the first parameter that had the NaN (with vcmp[UNORD_Q])
973
974 * Following pseudo code describes the algorithm for max[FD] (Min algorithm is on similar lines):
975 * btmp = (b < +0.0) ? a : b
976 * atmp = (b < +0.0) ? b : a
977 * Tmp = Max_Float(atmp , btmp)
978 * Res = (atmp == NaN) ? atmp : Tmp
979 */
980
981 void (MacroAssembler::*vblend)(XMMRegister, XMMRegister, XMMRegister, XMMRegister, int, bool, XMMRegister);
982 void (MacroAssembler::*vmaxmin)(XMMRegister, XMMRegister, XMMRegister, int);
983 void (MacroAssembler::*vcmp)(XMMRegister, XMMRegister, XMMRegister, int, int);
984 XMMRegister mask;
985
986 if (!is_double_word && is_min) {
987 mask = a;
988 vblend = &MacroAssembler::vblendvps;
989 vmaxmin = &MacroAssembler::vminps;
990 vcmp = &MacroAssembler::vcmpps;
991 } else if (!is_double_word && !is_min) {
992 mask = b;
993 vblend = &MacroAssembler::vblendvps;
994 vmaxmin = &MacroAssembler::vmaxps;
995 vcmp = &MacroAssembler::vcmpps;
996 } else if (is_double_word && is_min) {
997 mask = a;
998 vblend = &MacroAssembler::vblendvpd;
999 vmaxmin = &MacroAssembler::vminpd;
1000 vcmp = &MacroAssembler::vcmppd;
1001 } else {
1002 assert(is_double_word && !is_min, "sanity");
1003 mask = b;
1004 vblend = &MacroAssembler::vblendvpd;
1005 vmaxmin = &MacroAssembler::vmaxpd;
1006 vcmp = &MacroAssembler::vcmppd;
1007 }
1008
1009 // Make sure EnableX86ECoreOpts isn't disabled on register overlaps
1010 XMMRegister maxmin, scratch;
1011 if (dst == btmp) {
1012 maxmin = btmp;
1013 scratch = tmp;
1014 } else {
1015 maxmin = tmp;
1016 scratch = btmp;
1017 }
1018
1019 bool precompute_mask = EnableX86ECoreOpts && UseAVX>1;
1020 if (precompute_mask && !is_double_word) {
1021 vpsrad(tmp, mask, 32, vlen_enc);
1022 mask = tmp;
1023 } else if (precompute_mask && is_double_word) {
1024 vpxor(tmp, tmp, tmp, vlen_enc);
1025 vpcmpgtq(tmp, tmp, mask, vlen_enc);
1026 mask = tmp;
1027 }
1028
1029 (this->*vblend)(atmp, a, b, mask, vlen_enc, !precompute_mask, btmp);
1030 (this->*vblend)(btmp, b, a, mask, vlen_enc, !precompute_mask, tmp);
1031 (this->*vmaxmin)(maxmin, atmp, btmp, vlen_enc);
1032 (this->*vcmp)(scratch, atmp, atmp, Assembler::UNORD_Q, vlen_enc);
1033 (this->*vblend)(dst, maxmin, atmp, scratch, vlen_enc, false, scratch);
1034 }
1035
1036 void C2_MacroAssembler::evminmax_fp(int opcode, BasicType elem_bt,
1037 XMMRegister dst, XMMRegister a, XMMRegister b,
1038 KRegister ktmp, XMMRegister atmp, XMMRegister btmp,
1039 int vlen_enc) {
1040 assert(UseAVX > 2, "required");
1041 assert(opcode == Op_MinV || opcode == Op_MinReductionV ||
1042 opcode == Op_MaxV || opcode == Op_MaxReductionV, "sanity");
1043 assert(elem_bt == T_FLOAT || elem_bt == T_DOUBLE, "sanity");
1044 assert_different_registers(dst, a, atmp, btmp);
1045 assert_different_registers(dst, b, atmp, btmp);
1046
1047 bool is_min = (opcode == Op_MinV || opcode == Op_MinReductionV);
1048 bool is_double_word = is_double_word_type(elem_bt);
1049 bool merge = true;
1050
1051 if (!is_double_word && is_min) {
1052 evpmovd2m(ktmp, a, vlen_enc);
1053 evblendmps(atmp, ktmp, a, b, merge, vlen_enc);
1054 evblendmps(btmp, ktmp, b, a, merge, vlen_enc);
1055 vminps(dst, atmp, btmp, vlen_enc);
1056 evcmpps(ktmp, k0, atmp, atmp, Assembler::UNORD_Q, vlen_enc);
1057 evmovdqul(dst, ktmp, atmp, merge, vlen_enc);
1058 } else if (!is_double_word && !is_min) {
1059 evpmovd2m(ktmp, b, vlen_enc);
1060 evblendmps(atmp, ktmp, a, b, merge, vlen_enc);
1061 evblendmps(btmp, ktmp, b, a, merge, vlen_enc);
1062 vmaxps(dst, atmp, btmp, vlen_enc);
1063 evcmpps(ktmp, k0, atmp, atmp, Assembler::UNORD_Q, vlen_enc);
1064 evmovdqul(dst, ktmp, atmp, merge, vlen_enc);
1065 } else if (is_double_word && is_min) {
1066 evpmovq2m(ktmp, a, vlen_enc);
1067 evblendmpd(atmp, ktmp, a, b, merge, vlen_enc);
1068 evblendmpd(btmp, ktmp, b, a, merge, vlen_enc);
1069 vminpd(dst, atmp, btmp, vlen_enc);
1070 evcmppd(ktmp, k0, atmp, atmp, Assembler::UNORD_Q, vlen_enc);
1071 evmovdquq(dst, ktmp, atmp, merge, vlen_enc);
1072 } else {
1073 assert(is_double_word && !is_min, "sanity");
1074 evpmovq2m(ktmp, b, vlen_enc);
1075 evblendmpd(atmp, ktmp, a, b, merge, vlen_enc);
1076 evblendmpd(btmp, ktmp, b, a, merge, vlen_enc);
1077 vmaxpd(dst, atmp, btmp, vlen_enc);
1078 evcmppd(ktmp, k0, atmp, atmp, Assembler::UNORD_Q, vlen_enc);
1079 evmovdquq(dst, ktmp, atmp, merge, vlen_enc);
1080 }
1081 }
1082
1083 void C2_MacroAssembler::vminmax_fp_avx10_2(int opc, BasicType elem_bt, XMMRegister dst, KRegister mask,
1084 XMMRegister src1, XMMRegister src2, int vlen_enc) {
1085 assert(opc == Op_MinV || opc == Op_MinReductionV ||
1086 opc == Op_MaxV || opc == Op_MaxReductionV, "sanity");
1087
1088 int imm8 = (opc == Op_MinV || opc == Op_MinReductionV) ? AVX10_2_MINMAX_MIN_COMPARE_SIGN
1089 : AVX10_2_MINMAX_MAX_COMPARE_SIGN;
1090 if (elem_bt == T_FLOAT) {
1091 evminmaxps(dst, mask, src1, src2, true, imm8, vlen_enc);
1092 } else {
1093 assert(elem_bt == T_DOUBLE, "");
1094 evminmaxpd(dst, mask, src1, src2, true, imm8, vlen_enc);
1095 }
1096 }
1097
1098 void C2_MacroAssembler::sminmax_fp_avx10_2(int opc, BasicType elem_bt, XMMRegister dst, KRegister mask,
1099 XMMRegister src1, XMMRegister src2) {
1100 assert(opc == Op_MinF || opc == Op_MaxF ||
1101 opc == Op_MinD || opc == Op_MaxD, "sanity");
1102
1103 int imm8 = (opc == Op_MinF || opc == Op_MinD) ? AVX10_2_MINMAX_MIN_COMPARE_SIGN
1104 : AVX10_2_MINMAX_MAX_COMPARE_SIGN;
1105 if (elem_bt == T_FLOAT) {
1106 evminmaxss(dst, mask, src1, src2, true, imm8);
1107 } else {
1108 assert(elem_bt == T_DOUBLE, "");
1109 evminmaxsd(dst, mask, src1, src2, true, imm8);
1110 }
1111 }
1112
1113 // Float/Double signum
1114 void C2_MacroAssembler::signum_fp(int opcode, XMMRegister dst, XMMRegister zero, XMMRegister one) {
1115 assert(opcode == Op_SignumF || opcode == Op_SignumD, "sanity");
1116
1117 Label DONE_LABEL;
1118
1119 // Handle special cases +0.0/-0.0 and NaN, if argument is +0.0/-0.0 or NaN, return argument
1120 // If AVX10.2 (or newer) floating point comparison instructions used, SF=1 for equal and unordered cases
1121 // If other floating point comparison instructions used, ZF=1 for equal and unordered cases
1122 if (opcode == Op_SignumF) {
1123 if (VM_Version::supports_avx10_2()) {
1124 evucomxss(dst, zero);
1125 jcc(Assembler::negative, DONE_LABEL);
1126 } else {
1127 ucomiss(dst, zero);
1128 jcc(Assembler::equal, DONE_LABEL);
1129 }
1130 movflt(dst, one);
1131 jcc(Assembler::above, DONE_LABEL);
1132 xorps(dst, ExternalAddress(StubRoutines::x86::vector_float_sign_flip()), noreg);
1133 } else if (opcode == Op_SignumD) {
1134 if (VM_Version::supports_avx10_2()) {
1135 evucomxsd(dst, zero);
1136 jcc(Assembler::negative, DONE_LABEL);
1137 } else {
1138 ucomisd(dst, zero);
1139 jcc(Assembler::equal, DONE_LABEL);
1140 }
1141 movdbl(dst, one);
1142 jcc(Assembler::above, DONE_LABEL);
1143 xorpd(dst, ExternalAddress(StubRoutines::x86::vector_double_sign_flip()), noreg);
1144 }
1145
1146 bind(DONE_LABEL);
1147 }
1148
1149 void C2_MacroAssembler::vextendbw(bool sign, XMMRegister dst, XMMRegister src) {
1150 if (sign) {
1151 pmovsxbw(dst, src);
1152 } else {
1153 pmovzxbw(dst, src);
1154 }
1155 }
1156
1157 void C2_MacroAssembler::vextendbw(bool sign, XMMRegister dst, XMMRegister src, int vector_len) {
1158 if (sign) {
1159 vpmovsxbw(dst, src, vector_len);
1160 } else {
1161 vpmovzxbw(dst, src, vector_len);
1162 }
1163 }
1164
1165 void C2_MacroAssembler::vextendbd(bool sign, XMMRegister dst, XMMRegister src, int vector_len) {
1166 if (sign) {
1167 vpmovsxbd(dst, src, vector_len);
1168 } else {
1169 vpmovzxbd(dst, src, vector_len);
1170 }
1171 }
1172
1173 void C2_MacroAssembler::vextendwd(bool sign, XMMRegister dst, XMMRegister src, int vector_len) {
1174 if (sign) {
1175 vpmovsxwd(dst, src, vector_len);
1176 } else {
1177 vpmovzxwd(dst, src, vector_len);
1178 }
1179 }
1180
1181 void C2_MacroAssembler::vprotate_imm(int opcode, BasicType etype, XMMRegister dst, XMMRegister src,
1182 int shift, int vector_len) {
1183 if (opcode == Op_RotateLeftV) {
1184 if (etype == T_INT) {
1185 evprold(dst, src, shift, vector_len);
1186 } else {
1187 assert(etype == T_LONG, "expected type T_LONG");
1188 evprolq(dst, src, shift, vector_len);
1189 }
1190 } else {
1191 assert(opcode == Op_RotateRightV, "opcode should be Op_RotateRightV");
1192 if (etype == T_INT) {
1193 evprord(dst, src, shift, vector_len);
1194 } else {
1195 assert(etype == T_LONG, "expected type T_LONG");
1196 evprorq(dst, src, shift, vector_len);
1197 }
1198 }
1199 }
1200
1201 void C2_MacroAssembler::vprotate_var(int opcode, BasicType etype, XMMRegister dst, XMMRegister src,
1202 XMMRegister shift, int vector_len) {
1203 if (opcode == Op_RotateLeftV) {
1204 if (etype == T_INT) {
1205 evprolvd(dst, src, shift, vector_len);
1206 } else {
1207 assert(etype == T_LONG, "expected type T_LONG");
1208 evprolvq(dst, src, shift, vector_len);
1209 }
1210 } else {
1211 assert(opcode == Op_RotateRightV, "opcode should be Op_RotateRightV");
1212 if (etype == T_INT) {
1213 evprorvd(dst, src, shift, vector_len);
1214 } else {
1215 assert(etype == T_LONG, "expected type T_LONG");
1216 evprorvq(dst, src, shift, vector_len);
1217 }
1218 }
1219 }
1220
1221 void C2_MacroAssembler::vshiftd_imm(int opcode, XMMRegister dst, int shift) {
1222 if (opcode == Op_RShiftVI) {
1223 psrad(dst, shift);
1224 } else if (opcode == Op_LShiftVI) {
1225 pslld(dst, shift);
1226 } else {
1227 assert((opcode == Op_URShiftVI),"opcode should be Op_URShiftVI");
1228 psrld(dst, shift);
1229 }
1230 }
1231
1232 void C2_MacroAssembler::vshiftd(int opcode, XMMRegister dst, XMMRegister shift) {
1233 switch (opcode) {
1234 case Op_RShiftVI: psrad(dst, shift); break;
1235 case Op_LShiftVI: pslld(dst, shift); break;
1236 case Op_URShiftVI: psrld(dst, shift); break;
1237
1238 default: assert(false, "%s", NodeClassNames[opcode]);
1239 }
1240 }
1241
1242 void C2_MacroAssembler::vshiftd_imm(int opcode, XMMRegister dst, XMMRegister nds, int shift, int vector_len) {
1243 if (opcode == Op_RShiftVI) {
1244 vpsrad(dst, nds, shift, vector_len);
1245 } else if (opcode == Op_LShiftVI) {
1246 vpslld(dst, nds, shift, vector_len);
1247 } else {
1248 assert((opcode == Op_URShiftVI),"opcode should be Op_URShiftVI");
1249 vpsrld(dst, nds, shift, vector_len);
1250 }
1251 }
1252
1253 void C2_MacroAssembler::vshiftd(int opcode, XMMRegister dst, XMMRegister src, XMMRegister shift, int vlen_enc) {
1254 switch (opcode) {
1255 case Op_RShiftVI: vpsrad(dst, src, shift, vlen_enc); break;
1256 case Op_LShiftVI: vpslld(dst, src, shift, vlen_enc); break;
1257 case Op_URShiftVI: vpsrld(dst, src, shift, vlen_enc); break;
1258
1259 default: assert(false, "%s", NodeClassNames[opcode]);
1260 }
1261 }
1262
1263 void C2_MacroAssembler::vshiftw(int opcode, XMMRegister dst, XMMRegister shift) {
1264 switch (opcode) {
1265 case Op_RShiftVB: // fall-through
1266 case Op_RShiftVS: psraw(dst, shift); break;
1267
1268 case Op_LShiftVB: // fall-through
1269 case Op_LShiftVS: psllw(dst, shift); break;
1270
1271 case Op_URShiftVS: // fall-through
1272 case Op_URShiftVB: psrlw(dst, shift); break;
1273
1274 default: assert(false, "%s", NodeClassNames[opcode]);
1275 }
1276 }
1277
1278 void C2_MacroAssembler::vshiftw(int opcode, XMMRegister dst, XMMRegister src, XMMRegister shift, int vlen_enc) {
1279 switch (opcode) {
1280 case Op_RShiftVB: // fall-through
1281 case Op_RShiftVS: vpsraw(dst, src, shift, vlen_enc); break;
1282
1283 case Op_LShiftVB: // fall-through
1284 case Op_LShiftVS: vpsllw(dst, src, shift, vlen_enc); break;
1285
1286 case Op_URShiftVS: // fall-through
1287 case Op_URShiftVB: vpsrlw(dst, src, shift, vlen_enc); break;
1288
1289 default: assert(false, "%s", NodeClassNames[opcode]);
1290 }
1291 }
1292
1293 void C2_MacroAssembler::vshiftq(int opcode, XMMRegister dst, XMMRegister shift) {
1294 switch (opcode) {
1295 case Op_RShiftVL: psrlq(dst, shift); break; // using srl to implement sra on pre-avs512 systems
1296 case Op_LShiftVL: psllq(dst, shift); break;
1297 case Op_URShiftVL: psrlq(dst, shift); break;
1298
1299 default: assert(false, "%s", NodeClassNames[opcode]);
1300 }
1301 }
1302
1303 void C2_MacroAssembler::vshiftq_imm(int opcode, XMMRegister dst, int shift) {
1304 if (opcode == Op_RShiftVL) {
1305 psrlq(dst, shift); // using srl to implement sra on pre-avs512 systems
1306 } else if (opcode == Op_LShiftVL) {
1307 psllq(dst, shift);
1308 } else {
1309 assert((opcode == Op_URShiftVL),"opcode should be Op_URShiftVL");
1310 psrlq(dst, shift);
1311 }
1312 }
1313
1314 void C2_MacroAssembler::vshiftq(int opcode, XMMRegister dst, XMMRegister src, XMMRegister shift, int vlen_enc) {
1315 switch (opcode) {
1316 case Op_RShiftVL: evpsraq(dst, src, shift, vlen_enc); break;
1317 case Op_LShiftVL: vpsllq(dst, src, shift, vlen_enc); break;
1318 case Op_URShiftVL: vpsrlq(dst, src, shift, vlen_enc); break;
1319
1320 default: assert(false, "%s", NodeClassNames[opcode]);
1321 }
1322 }
1323
1324 void C2_MacroAssembler::vshiftq_imm(int opcode, XMMRegister dst, XMMRegister nds, int shift, int vector_len) {
1325 if (opcode == Op_RShiftVL) {
1326 evpsraq(dst, nds, shift, vector_len);
1327 } else if (opcode == Op_LShiftVL) {
1328 vpsllq(dst, nds, shift, vector_len);
1329 } else {
1330 assert((opcode == Op_URShiftVL),"opcode should be Op_URShiftVL");
1331 vpsrlq(dst, nds, shift, vector_len);
1332 }
1333 }
1334
1335 void C2_MacroAssembler::varshiftd(int opcode, XMMRegister dst, XMMRegister src, XMMRegister shift, int vlen_enc) {
1336 switch (opcode) {
1337 case Op_RShiftVB: // fall-through
1338 case Op_RShiftVS: // fall-through
1339 case Op_RShiftVI: vpsravd(dst, src, shift, vlen_enc); break;
1340
1341 case Op_LShiftVB: // fall-through
1342 case Op_LShiftVS: // fall-through
1343 case Op_LShiftVI: vpsllvd(dst, src, shift, vlen_enc); break;
1344
1345 case Op_URShiftVB: // fall-through
1346 case Op_URShiftVS: // fall-through
1347 case Op_URShiftVI: vpsrlvd(dst, src, shift, vlen_enc); break;
1348
1349 default: assert(false, "%s", NodeClassNames[opcode]);
1350 }
1351 }
1352
1353 void C2_MacroAssembler::varshiftw(int opcode, XMMRegister dst, XMMRegister src, XMMRegister shift, int vlen_enc) {
1354 switch (opcode) {
1355 case Op_RShiftVB: // fall-through
1356 case Op_RShiftVS: evpsravw(dst, src, shift, vlen_enc); break;
1357
1358 case Op_LShiftVB: // fall-through
1359 case Op_LShiftVS: evpsllvw(dst, src, shift, vlen_enc); break;
1360
1361 case Op_URShiftVB: // fall-through
1362 case Op_URShiftVS: evpsrlvw(dst, src, shift, vlen_enc); break;
1363
1364 default: assert(false, "%s", NodeClassNames[opcode]);
1365 }
1366 }
1367
1368 void C2_MacroAssembler::varshiftq(int opcode, XMMRegister dst, XMMRegister src, XMMRegister shift, int vlen_enc, XMMRegister tmp) {
1369 assert(UseAVX >= 2, "required");
1370 switch (opcode) {
1371 case Op_RShiftVL: {
1372 if (UseAVX > 2) {
1373 assert(tmp == xnoreg, "not used");
1374 if (!VM_Version::supports_avx512vl()) {
1375 vlen_enc = Assembler::AVX_512bit;
1376 }
1377 evpsravq(dst, src, shift, vlen_enc);
1378 } else {
1379 vmovdqu(tmp, ExternalAddress(StubRoutines::x86::vector_long_sign_mask()));
1380 vpsrlvq(dst, src, shift, vlen_enc);
1381 vpsrlvq(tmp, tmp, shift, vlen_enc);
1382 vpxor(dst, dst, tmp, vlen_enc);
1383 vpsubq(dst, dst, tmp, vlen_enc);
1384 }
1385 break;
1386 }
1387 case Op_LShiftVL: {
1388 assert(tmp == xnoreg, "not used");
1389 vpsllvq(dst, src, shift, vlen_enc);
1390 break;
1391 }
1392 case Op_URShiftVL: {
1393 assert(tmp == xnoreg, "not used");
1394 vpsrlvq(dst, src, shift, vlen_enc);
1395 break;
1396 }
1397 default: assert(false, "%s", NodeClassNames[opcode]);
1398 }
1399 }
1400
1401 // Variable shift src by shift using vtmp and scratch as TEMPs giving word result in dst
1402 void C2_MacroAssembler::varshiftbw(int opcode, XMMRegister dst, XMMRegister src, XMMRegister shift, int vector_len, XMMRegister vtmp) {
1403 assert(opcode == Op_LShiftVB ||
1404 opcode == Op_RShiftVB ||
1405 opcode == Op_URShiftVB, "%s", NodeClassNames[opcode]);
1406 bool sign = (opcode != Op_URShiftVB);
1407 assert(vector_len == 0, "required");
1408 vextendbd(sign, dst, src, 1);
1409 vpmovzxbd(vtmp, shift, 1);
1410 varshiftd(opcode, dst, dst, vtmp, 1);
1411 vpand(dst, dst, ExternalAddress(StubRoutines::x86::vector_int_to_byte_mask()), 1, noreg);
1412 vextracti128_high(vtmp, dst);
1413 vpackusdw(dst, dst, vtmp, 0);
1414 }
1415
1416 // Variable shift src by shift using vtmp and scratch as TEMPs giving byte result in dst
1417 void C2_MacroAssembler::evarshiftb(int opcode, XMMRegister dst, XMMRegister src, XMMRegister shift, int vector_len, XMMRegister vtmp) {
1418 assert(opcode == Op_LShiftVB ||
1419 opcode == Op_RShiftVB ||
1420 opcode == Op_URShiftVB, "%s", NodeClassNames[opcode]);
1421 bool sign = (opcode != Op_URShiftVB);
1422 int ext_vector_len = vector_len + 1;
1423 vextendbw(sign, dst, src, ext_vector_len);
1424 vpmovzxbw(vtmp, shift, ext_vector_len);
1425 varshiftw(opcode, dst, dst, vtmp, ext_vector_len);
1426 vpand(dst, dst, ExternalAddress(StubRoutines::x86::vector_short_to_byte_mask()), ext_vector_len, noreg);
1427 if (vector_len == 0) {
1428 vextracti128_high(vtmp, dst);
1429 vpackuswb(dst, dst, vtmp, vector_len);
1430 } else {
1431 vextracti64x4_high(vtmp, dst);
1432 vpackuswb(dst, dst, vtmp, vector_len);
1433 vpermq(dst, dst, 0xD8, vector_len);
1434 }
1435 }
1436
1437 void C2_MacroAssembler::insert(BasicType typ, XMMRegister dst, Register val, int idx) {
1438 switch(typ) {
1439 case T_BYTE:
1440 pinsrb(dst, val, idx);
1441 break;
1442 case T_SHORT:
1443 pinsrw(dst, val, idx);
1444 break;
1445 case T_INT:
1446 pinsrd(dst, val, idx);
1447 break;
1448 case T_LONG:
1449 pinsrq(dst, val, idx);
1450 break;
1451 default:
1452 assert(false,"Should not reach here.");
1453 break;
1454 }
1455 }
1456
1457 void C2_MacroAssembler::vinsert(BasicType typ, XMMRegister dst, XMMRegister src, Register val, int idx) {
1458 switch(typ) {
1459 case T_BYTE:
1460 vpinsrb(dst, src, val, idx);
1461 break;
1462 case T_SHORT:
1463 vpinsrw(dst, src, val, idx);
1464 break;
1465 case T_INT:
1466 vpinsrd(dst, src, val, idx);
1467 break;
1468 case T_LONG:
1469 vpinsrq(dst, src, val, idx);
1470 break;
1471 default:
1472 assert(false,"Should not reach here.");
1473 break;
1474 }
1475 }
1476
1477 void C2_MacroAssembler::vgather8b_masked(BasicType elem_bt, XMMRegister dst,
1478 Register base, Register idx_base,
1479 Register mask, Register mask_idx,
1480 Register rtmp, int vlen_enc) {
1481 vpxor(dst, dst, dst, vlen_enc);
1482 if (elem_bt == T_SHORT) {
1483 for (int i = 0; i < 4; i++) {
1484 // dst[i] = mask[i] ? src[idx_base[i]] : 0
1485 Label skip_load;
1486 btq(mask, mask_idx);
1487 jccb(Assembler::carryClear, skip_load);
1488 movl(rtmp, Address(idx_base, i * 4));
1489 pinsrw(dst, Address(base, rtmp, Address::times_2), i);
1490 bind(skip_load);
1491 incq(mask_idx);
1492 }
1493 } else {
1494 assert(elem_bt == T_BYTE, "");
1495 for (int i = 0; i < 8; i++) {
1496 // dst[i] = mask[i] ? src[idx_base[i]] : 0
1497 Label skip_load;
1498 btq(mask, mask_idx);
1499 jccb(Assembler::carryClear, skip_load);
1500 movl(rtmp, Address(idx_base, i * 4));
1501 pinsrb(dst, Address(base, rtmp), i);
1502 bind(skip_load);
1503 incq(mask_idx);
1504 }
1505 }
1506 }
1507
1508 void C2_MacroAssembler::vgather8b(BasicType elem_bt, XMMRegister dst,
1509 Register base, Register idx_base,
1510 Register rtmp, int vlen_enc) {
1511 vpxor(dst, dst, dst, vlen_enc);
1512 if (elem_bt == T_SHORT) {
1513 for (int i = 0; i < 4; i++) {
1514 // dst[i] = src[idx_base[i]]
1515 movl(rtmp, Address(idx_base, i * 4));
1516 pinsrw(dst, Address(base, rtmp, Address::times_2), i);
1517 }
1518 } else {
1519 assert(elem_bt == T_BYTE, "");
1520 for (int i = 0; i < 8; i++) {
1521 // dst[i] = src[idx_base[i]]
1522 movl(rtmp, Address(idx_base, i * 4));
1523 pinsrb(dst, Address(base, rtmp), i);
1524 }
1525 }
1526 }
1527
1528 /*
1529 * Gather using hybrid algorithm, first partially unroll scalar loop
1530 * to accumulate values from gather indices into a quad-word(64bit) slice.
1531 * A slice may hold 8 bytes or 4 short values. This is followed by a vector
1532 * permutation to place the slice into appropriate vector lane
1533 * locations in destination vector. Following pseudo code describes the
1534 * algorithm in detail:
1535 *
1536 * DST_VEC = ZERO_VEC
1537 * PERM_INDEX = {0, 1, 2, 3, 4, 5, 6, 7, 8..}
1538 * TWO_VEC = {2, 2, 2, 2, 2, 2, 2, 2, 2..}
1539 * FOREACH_ITER:
1540 * TMP_VEC_64 = PICK_SUB_WORDS_FROM_GATHER_INDICES
1541 * TEMP_PERM_VEC = PERMUTE TMP_VEC_64 PERM_INDEX
1542 * DST_VEC = DST_VEC OR TEMP_PERM_VEC
1543 * PERM_INDEX = PERM_INDEX - TWO_VEC
1544 *
1545 * With each iteration, doubleword permute indices (0,1) corresponding
1546 * to gathered quadword gets right shifted by two lane positions.
1547 *
1548 */
1549 void C2_MacroAssembler::vgather_subword(BasicType elem_ty, XMMRegister dst,
1550 Register base, Register idx_base,
1551 Register mask, XMMRegister xtmp1,
1552 XMMRegister xtmp2, XMMRegister temp_dst,
1553 Register rtmp, Register mask_idx,
1554 Register length, int vector_len, int vlen_enc) {
1555 Label GATHER8_LOOP;
1556 assert(is_subword_type(elem_ty), "");
1557 movl(length, vector_len);
1558 vpxor(xtmp1, xtmp1, xtmp1, vlen_enc); // xtmp1 = {0, ...}
1559 vpxor(dst, dst, dst, vlen_enc); // dst = {0, ...}
1560 vallones(xtmp2, vlen_enc);
1561 vpsubd(xtmp2, xtmp1, xtmp2, vlen_enc);
1562 vpslld(xtmp2, xtmp2, 1, vlen_enc); // xtmp2 = {2, 2, ...}
1563 load_iota_indices(xtmp1, vector_len * type2aelembytes(elem_ty), T_INT); // xtmp1 = {0, 1, 2, ...}
1564
1565 bind(GATHER8_LOOP);
1566 // TMP_VEC_64(temp_dst) = PICK_SUB_WORDS_FROM_GATHER_INDICES
1567 if (mask == noreg) {
1568 vgather8b(elem_ty, temp_dst, base, idx_base, rtmp, vlen_enc);
1569 } else {
1570 vgather8b_masked(elem_ty, temp_dst, base, idx_base, mask, mask_idx, rtmp, vlen_enc);
1571 }
1572 // TEMP_PERM_VEC(temp_dst) = PERMUTE TMP_VEC_64(temp_dst) PERM_INDEX(xtmp1)
1573 vpermd(temp_dst, xtmp1, temp_dst, vlen_enc == Assembler::AVX_512bit ? vlen_enc : Assembler::AVX_256bit);
1574 // PERM_INDEX(xtmp1) = PERM_INDEX(xtmp1) - TWO_VEC(xtmp2)
1575 vpsubd(xtmp1, xtmp1, xtmp2, vlen_enc);
1576 // DST_VEC = DST_VEC OR TEMP_PERM_VEC
1577 vpor(dst, dst, temp_dst, vlen_enc);
1578 addptr(idx_base, 32 >> (type2aelembytes(elem_ty) - 1));
1579 subl(length, 8 >> (type2aelembytes(elem_ty) - 1));
1580 jcc(Assembler::notEqual, GATHER8_LOOP);
1581 }
1582
1583 void C2_MacroAssembler::vgather(BasicType typ, XMMRegister dst, Register base, XMMRegister idx, XMMRegister mask, int vector_len) {
1584 switch(typ) {
1585 case T_INT:
1586 vpgatherdd(dst, Address(base, idx, Address::times_4), mask, vector_len);
1587 break;
1588 case T_FLOAT:
1589 vgatherdps(dst, Address(base, idx, Address::times_4), mask, vector_len);
1590 break;
1591 case T_LONG:
1592 vpgatherdq(dst, Address(base, idx, Address::times_8), mask, vector_len);
1593 break;
1594 case T_DOUBLE:
1595 vgatherdpd(dst, Address(base, idx, Address::times_8), mask, vector_len);
1596 break;
1597 default:
1598 assert(false,"Should not reach here.");
1599 break;
1600 }
1601 }
1602
1603 void C2_MacroAssembler::evgather(BasicType typ, XMMRegister dst, KRegister mask, Register base, XMMRegister idx, int vector_len) {
1604 switch(typ) {
1605 case T_INT:
1606 evpgatherdd(dst, mask, Address(base, idx, Address::times_4), vector_len);
1607 break;
1608 case T_FLOAT:
1609 evgatherdps(dst, mask, Address(base, idx, Address::times_4), vector_len);
1610 break;
1611 case T_LONG:
1612 evpgatherdq(dst, mask, Address(base, idx, Address::times_8), vector_len);
1613 break;
1614 case T_DOUBLE:
1615 evgatherdpd(dst, mask, Address(base, idx, Address::times_8), vector_len);
1616 break;
1617 default:
1618 assert(false,"Should not reach here.");
1619 break;
1620 }
1621 }
1622
1623 void C2_MacroAssembler::evscatter(BasicType typ, Register base, XMMRegister idx, KRegister mask, XMMRegister src, int vector_len) {
1624 switch(typ) {
1625 case T_INT:
1626 evpscatterdd(Address(base, idx, Address::times_4), mask, src, vector_len);
1627 break;
1628 case T_FLOAT:
1629 evscatterdps(Address(base, idx, Address::times_4), mask, src, vector_len);
1630 break;
1631 case T_LONG:
1632 evpscatterdq(Address(base, idx, Address::times_8), mask, src, vector_len);
1633 break;
1634 case T_DOUBLE:
1635 evscatterdpd(Address(base, idx, Address::times_8), mask, src, vector_len);
1636 break;
1637 default:
1638 assert(false,"Should not reach here.");
1639 break;
1640 }
1641 }
1642
1643 void C2_MacroAssembler::load_vector_mask(XMMRegister dst, XMMRegister src, int vlen_in_bytes, BasicType elem_bt, bool is_legacy) {
1644 if (vlen_in_bytes <= 16) {
1645 pxor (dst, dst);
1646 psubb(dst, src);
1647 switch (elem_bt) {
1648 case T_BYTE: /* nothing to do */ break;
1649 case T_SHORT: pmovsxbw(dst, dst); break;
1650 case T_INT: pmovsxbd(dst, dst); break;
1651 case T_FLOAT: pmovsxbd(dst, dst); break;
1652 case T_LONG: pmovsxbq(dst, dst); break;
1653 case T_DOUBLE: pmovsxbq(dst, dst); break;
1654
1655 default: assert(false, "%s", type2name(elem_bt));
1656 }
1657 } else {
1658 assert(!is_legacy || !is_subword_type(elem_bt) || vlen_in_bytes < 64, "");
1659 int vlen_enc = vector_length_encoding(vlen_in_bytes);
1660
1661 vpxor (dst, dst, dst, vlen_enc);
1662 vpsubb(dst, dst, src, is_legacy ? AVX_256bit : vlen_enc);
1663
1664 switch (elem_bt) {
1665 case T_BYTE: /* nothing to do */ break;
1666 case T_SHORT: vpmovsxbw(dst, dst, vlen_enc); break;
1667 case T_INT: vpmovsxbd(dst, dst, vlen_enc); break;
1668 case T_FLOAT: vpmovsxbd(dst, dst, vlen_enc); break;
1669 case T_LONG: vpmovsxbq(dst, dst, vlen_enc); break;
1670 case T_DOUBLE: vpmovsxbq(dst, dst, vlen_enc); break;
1671
1672 default: assert(false, "%s", type2name(elem_bt));
1673 }
1674 }
1675 }
1676
1677 void C2_MacroAssembler::load_vector_mask(KRegister dst, XMMRegister src, XMMRegister xtmp, bool novlbwdq, int vlen_enc) {
1678 if (novlbwdq) {
1679 vpmovsxbd(xtmp, src, vlen_enc);
1680 evpcmpd(dst, k0, xtmp, ExternalAddress(StubRoutines::x86::vector_int_mask_cmp_bits()),
1681 Assembler::eq, true, vlen_enc, noreg);
1682 } else {
1683 vpxor(xtmp, xtmp, xtmp, vlen_enc);
1684 vpsubb(xtmp, xtmp, src, vlen_enc);
1685 evpmovb2m(dst, xtmp, vlen_enc);
1686 }
1687 }
1688
1689 void C2_MacroAssembler::load_vector(BasicType bt, XMMRegister dst, Address src, int vlen_in_bytes) {
1690 if (is_integral_type(bt)) {
1691 switch (vlen_in_bytes) {
1692 case 4: movdl(dst, src); break;
1693 case 8: movq(dst, src); break;
1694 case 16: movdqu(dst, src); break;
1695 case 32: vmovdqu(dst, src); break;
1696 case 64: evmovdqul(dst, src, Assembler::AVX_512bit); break;
1697 default: ShouldNotReachHere();
1698 }
1699 } else {
1700 switch (vlen_in_bytes) {
1701 case 4: movflt(dst, src); break;
1702 case 8: movdbl(dst, src); break;
1703 case 16: movups(dst, src); break;
1704 case 32: vmovups(dst, src, Assembler::AVX_256bit); break;
1705 case 64: vmovups(dst, src, Assembler::AVX_512bit); break;
1706 default: ShouldNotReachHere();
1707 }
1708 }
1709 }
1710
1711 void C2_MacroAssembler::load_vector(BasicType bt, XMMRegister dst, AddressLiteral src, int vlen_in_bytes, Register rscratch) {
1712 assert(rscratch != noreg || always_reachable(src), "missing");
1713
1714 if (reachable(src)) {
1715 load_vector(bt, dst, as_Address(src), vlen_in_bytes);
1716 } else {
1717 lea(rscratch, src);
1718 load_vector(bt, dst, Address(rscratch, 0), vlen_in_bytes);
1719 }
1720 }
1721
1722 void C2_MacroAssembler::load_constant_vector(BasicType bt, XMMRegister dst, InternalAddress src, int vlen) {
1723 int vlen_enc = vector_length_encoding(vlen);
1724 if (VM_Version::supports_avx()) {
1725 if (bt == T_LONG) {
1726 if (VM_Version::supports_avx2()) {
1727 vpbroadcastq(dst, src, vlen_enc);
1728 } else {
1729 vmovddup(dst, src, vlen_enc);
1730 }
1731 } else if (bt == T_DOUBLE) {
1732 if (vlen_enc != Assembler::AVX_128bit) {
1733 vbroadcastsd(dst, src, vlen_enc, noreg);
1734 } else {
1735 vmovddup(dst, src, vlen_enc);
1736 }
1737 } else {
1738 if (VM_Version::supports_avx2() && is_integral_type(bt)) {
1739 vpbroadcastd(dst, src, vlen_enc);
1740 } else {
1741 vbroadcastss(dst, src, vlen_enc);
1742 }
1743 }
1744 } else if (VM_Version::supports_sse3()) {
1745 movddup(dst, src);
1746 } else {
1747 load_vector(bt, dst, src, vlen);
1748 }
1749 }
1750
1751 void C2_MacroAssembler::load_iota_indices(XMMRegister dst, int vlen_in_bytes, BasicType bt) {
1752 int entry_idx = vector_iota_entry_index(bt);
1753 ExternalAddress addr(StubRoutines::x86::vector_iota_indices(entry_idx));
1754 load_vector(T_BYTE, dst, addr, vlen_in_bytes);
1755 }
1756
1757 // Reductions for vectors of bytes, shorts, ints, longs, floats, and doubles.
1758
1759 void C2_MacroAssembler::reduce_operation_128(BasicType typ, int opcode, XMMRegister dst, XMMRegister src) {
1760 int vector_len = Assembler::AVX_128bit;
1761
1762 switch (opcode) {
1763 case Op_AndReductionV: pand(dst, src); break;
1764 case Op_OrReductionV: por (dst, src); break;
1765 case Op_XorReductionV: pxor(dst, src); break;
1766 case Op_MinReductionV:
1767 switch (typ) {
1768 case T_BYTE: pminsb(dst, src); break;
1769 case T_SHORT: pminsw(dst, src); break;
1770 case T_INT: pminsd(dst, src); break;
1771 case T_LONG: assert(UseAVX > 2, "required");
1772 vpminsq(dst, dst, src, Assembler::AVX_128bit); break;
1773 default: assert(false, "wrong type");
1774 }
1775 break;
1776 case Op_MaxReductionV:
1777 switch (typ) {
1778 case T_BYTE: pmaxsb(dst, src); break;
1779 case T_SHORT: pmaxsw(dst, src); break;
1780 case T_INT: pmaxsd(dst, src); break;
1781 case T_LONG: assert(UseAVX > 2, "required");
1782 vpmaxsq(dst, dst, src, Assembler::AVX_128bit); break;
1783 default: assert(false, "wrong type");
1784 }
1785 break;
1786 case Op_UMinReductionV:
1787 switch (typ) {
1788 case T_BYTE: vpminub(dst, dst, src, Assembler::AVX_128bit); break;
1789 case T_SHORT: vpminuw(dst, dst, src, Assembler::AVX_128bit); break;
1790 case T_INT: vpminud(dst, dst, src, Assembler::AVX_128bit); break;
1791 case T_LONG: evpminuq(dst, k0, dst, src, true, Assembler::AVX_128bit); break;
1792 default: assert(false, "wrong type");
1793 }
1794 break;
1795 case Op_UMaxReductionV:
1796 switch (typ) {
1797 case T_BYTE: vpmaxub(dst, dst, src, Assembler::AVX_128bit); break;
1798 case T_SHORT: vpmaxuw(dst, dst, src, Assembler::AVX_128bit); break;
1799 case T_INT: vpmaxud(dst, dst, src, Assembler::AVX_128bit); break;
1800 case T_LONG: evpmaxuq(dst, k0, dst, src, true, Assembler::AVX_128bit); break;
1801 default: assert(false, "wrong type");
1802 }
1803 break;
1804 case Op_AddReductionVF: addss(dst, src); break;
1805 case Op_AddReductionVD: addsd(dst, src); break;
1806 case Op_AddReductionVI:
1807 switch (typ) {
1808 case T_BYTE: paddb(dst, src); break;
1809 case T_SHORT: paddw(dst, src); break;
1810 case T_INT: paddd(dst, src); break;
1811 default: assert(false, "wrong type");
1812 }
1813 break;
1814 case Op_AddReductionVL: paddq(dst, src); break;
1815 case Op_MulReductionVF: mulss(dst, src); break;
1816 case Op_MulReductionVD: mulsd(dst, src); break;
1817 case Op_MulReductionVI:
1818 switch (typ) {
1819 case T_SHORT: pmullw(dst, src); break;
1820 case T_INT: pmulld(dst, src); break;
1821 default: assert(false, "wrong type");
1822 }
1823 break;
1824 case Op_MulReductionVL: assert(UseAVX > 2, "required");
1825 evpmullq(dst, dst, src, vector_len); break;
1826 default: assert(false, "wrong opcode");
1827 }
1828 }
1829
1830 void C2_MacroAssembler::unordered_reduce_operation_128(BasicType typ, int opcode, XMMRegister dst, XMMRegister src) {
1831 switch (opcode) {
1832 case Op_AddReductionVF: addps(dst, src); break;
1833 case Op_AddReductionVD: addpd(dst, src); break;
1834 case Op_MulReductionVF: mulps(dst, src); break;
1835 case Op_MulReductionVD: mulpd(dst, src); break;
1836 default: assert(false, "%s", NodeClassNames[opcode]);
1837 }
1838 }
1839
1840 void C2_MacroAssembler::reduce_operation_256(BasicType typ, int opcode, XMMRegister dst, XMMRegister src1, XMMRegister src2) {
1841 int vector_len = Assembler::AVX_256bit;
1842
1843 switch (opcode) {
1844 case Op_AndReductionV: vpand(dst, src1, src2, vector_len); break;
1845 case Op_OrReductionV: vpor (dst, src1, src2, vector_len); break;
1846 case Op_XorReductionV: vpxor(dst, src1, src2, vector_len); break;
1847 case Op_MinReductionV:
1848 switch (typ) {
1849 case T_BYTE: vpminsb(dst, src1, src2, vector_len); break;
1850 case T_SHORT: vpminsw(dst, src1, src2, vector_len); break;
1851 case T_INT: vpminsd(dst, src1, src2, vector_len); break;
1852 case T_LONG: assert(UseAVX > 2, "required");
1853 vpminsq(dst, src1, src2, vector_len); break;
1854 default: assert(false, "wrong type");
1855 }
1856 break;
1857 case Op_MaxReductionV:
1858 switch (typ) {
1859 case T_BYTE: vpmaxsb(dst, src1, src2, vector_len); break;
1860 case T_SHORT: vpmaxsw(dst, src1, src2, vector_len); break;
1861 case T_INT: vpmaxsd(dst, src1, src2, vector_len); break;
1862 case T_LONG: assert(UseAVX > 2, "required");
1863 vpmaxsq(dst, src1, src2, vector_len); break;
1864 default: assert(false, "wrong type");
1865 }
1866 break;
1867 case Op_UMinReductionV:
1868 switch (typ) {
1869 case T_BYTE: vpminub(dst, src1, src2, vector_len); break;
1870 case T_SHORT: vpminuw(dst, src1, src2, vector_len); break;
1871 case T_INT: vpminud(dst, src1, src2, vector_len); break;
1872 case T_LONG: evpminuq(dst, k0, src1, src2, true, vector_len); break;
1873 default: assert(false, "wrong type");
1874 }
1875 break;
1876 case Op_UMaxReductionV:
1877 switch (typ) {
1878 case T_BYTE: vpmaxub(dst, src1, src2, vector_len); break;
1879 case T_SHORT: vpmaxuw(dst, src1, src2, vector_len); break;
1880 case T_INT: vpmaxud(dst, src1, src2, vector_len); break;
1881 case T_LONG: evpmaxuq(dst, k0, src1, src2, true, vector_len); break;
1882 default: assert(false, "wrong type");
1883 }
1884 break;
1885 case Op_AddReductionVI:
1886 switch (typ) {
1887 case T_BYTE: vpaddb(dst, src1, src2, vector_len); break;
1888 case T_SHORT: vpaddw(dst, src1, src2, vector_len); break;
1889 case T_INT: vpaddd(dst, src1, src2, vector_len); break;
1890 default: assert(false, "wrong type");
1891 }
1892 break;
1893 case Op_AddReductionVL: vpaddq(dst, src1, src2, vector_len); break;
1894 case Op_MulReductionVI:
1895 switch (typ) {
1896 case T_SHORT: vpmullw(dst, src1, src2, vector_len); break;
1897 case T_INT: vpmulld(dst, src1, src2, vector_len); break;
1898 default: assert(false, "wrong type");
1899 }
1900 break;
1901 case Op_MulReductionVL: evpmullq(dst, src1, src2, vector_len); break;
1902 default: assert(false, "wrong opcode");
1903 }
1904 }
1905
1906 void C2_MacroAssembler::unordered_reduce_operation_256(BasicType typ, int opcode, XMMRegister dst, XMMRegister src1, XMMRegister src2) {
1907 int vector_len = Assembler::AVX_256bit;
1908
1909 switch (opcode) {
1910 case Op_AddReductionVF: vaddps(dst, src1, src2, vector_len); break;
1911 case Op_AddReductionVD: vaddpd(dst, src1, src2, vector_len); break;
1912 case Op_MulReductionVF: vmulps(dst, src1, src2, vector_len); break;
1913 case Op_MulReductionVD: vmulpd(dst, src1, src2, vector_len); break;
1914 default: assert(false, "%s", NodeClassNames[opcode]);
1915 }
1916 }
1917
1918 void C2_MacroAssembler::reduce_fp(int opcode, int vlen,
1919 XMMRegister dst, XMMRegister src,
1920 XMMRegister vtmp1, XMMRegister vtmp2) {
1921 switch (opcode) {
1922 case Op_AddReductionVF:
1923 case Op_MulReductionVF:
1924 reduceF(opcode, vlen, dst, src, vtmp1, vtmp2);
1925 break;
1926
1927 case Op_AddReductionVD:
1928 case Op_MulReductionVD:
1929 reduceD(opcode, vlen, dst, src, vtmp1, vtmp2);
1930 break;
1931
1932 default: assert(false, "wrong opcode");
1933 }
1934 }
1935
1936 void C2_MacroAssembler::unordered_reduce_fp(int opcode, int vlen,
1937 XMMRegister dst, XMMRegister src,
1938 XMMRegister vtmp1, XMMRegister vtmp2) {
1939 switch (opcode) {
1940 case Op_AddReductionVF:
1941 case Op_MulReductionVF:
1942 unorderedReduceF(opcode, vlen, dst, src, vtmp1, vtmp2);
1943 break;
1944
1945 case Op_AddReductionVD:
1946 case Op_MulReductionVD:
1947 unorderedReduceD(opcode, vlen, dst, src, vtmp1, vtmp2);
1948 break;
1949
1950 default: assert(false, "%s", NodeClassNames[opcode]);
1951 }
1952 }
1953
1954 void C2_MacroAssembler::reduceB(int opcode, int vlen,
1955 Register dst, Register src1, XMMRegister src2,
1956 XMMRegister vtmp1, XMMRegister vtmp2) {
1957 switch (vlen) {
1958 case 8: reduce8B (opcode, dst, src1, src2, vtmp1, vtmp2); break;
1959 case 16: reduce16B(opcode, dst, src1, src2, vtmp1, vtmp2); break;
1960 case 32: reduce32B(opcode, dst, src1, src2, vtmp1, vtmp2); break;
1961 case 64: reduce64B(opcode, dst, src1, src2, vtmp1, vtmp2); break;
1962
1963 default: assert(false, "wrong vector length");
1964 }
1965 }
1966
1967 void C2_MacroAssembler::mulreduceB(int opcode, int vlen,
1968 Register dst, Register src1, XMMRegister src2,
1969 XMMRegister vtmp1, XMMRegister vtmp2) {
1970 switch (vlen) {
1971 case 8: mulreduce8B (opcode, dst, src1, src2, vtmp1, vtmp2); break;
1972 case 16: mulreduce16B(opcode, dst, src1, src2, vtmp1, vtmp2); break;
1973 case 32: mulreduce32B(opcode, dst, src1, src2, vtmp1, vtmp2); break;
1974 case 64: mulreduce64B(opcode, dst, src1, src2, vtmp1, vtmp2); break;
1975
1976 default: assert(false, "wrong vector length");
1977 }
1978 }
1979
1980 void C2_MacroAssembler::reduceS(int opcode, int vlen,
1981 Register dst, Register src1, XMMRegister src2,
1982 XMMRegister vtmp1, XMMRegister vtmp2) {
1983 switch (vlen) {
1984 case 4: reduce4S (opcode, dst, src1, src2, vtmp1, vtmp2); break;
1985 case 8: reduce8S (opcode, dst, src1, src2, vtmp1, vtmp2); break;
1986 case 16: reduce16S(opcode, dst, src1, src2, vtmp1, vtmp2); break;
1987 case 32: reduce32S(opcode, dst, src1, src2, vtmp1, vtmp2); break;
1988
1989 default: assert(false, "wrong vector length");
1990 }
1991 }
1992
1993 void C2_MacroAssembler::reduceI(int opcode, int vlen,
1994 Register dst, Register src1, XMMRegister src2,
1995 XMMRegister vtmp1, XMMRegister vtmp2) {
1996 switch (vlen) {
1997 case 2: reduce2I (opcode, dst, src1, src2, vtmp1, vtmp2); break;
1998 case 4: reduce4I (opcode, dst, src1, src2, vtmp1, vtmp2); break;
1999 case 8: reduce8I (opcode, dst, src1, src2, vtmp1, vtmp2); break;
2000 case 16: reduce16I(opcode, dst, src1, src2, vtmp1, vtmp2); break;
2001
2002 default: assert(false, "wrong vector length");
2003 }
2004 }
2005
2006 void C2_MacroAssembler::reduceL(int opcode, int vlen,
2007 Register dst, Register src1, XMMRegister src2,
2008 XMMRegister vtmp1, XMMRegister vtmp2) {
2009 switch (vlen) {
2010 case 2: reduce2L(opcode, dst, src1, src2, vtmp1, vtmp2); break;
2011 case 4: reduce4L(opcode, dst, src1, src2, vtmp1, vtmp2); break;
2012 case 8: reduce8L(opcode, dst, src1, src2, vtmp1, vtmp2); break;
2013
2014 default: assert(false, "wrong vector length");
2015 }
2016 }
2017
2018 void C2_MacroAssembler::reduceF(int opcode, int vlen, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2019 switch (vlen) {
2020 case 2:
2021 assert(vtmp2 == xnoreg, "");
2022 reduce2F(opcode, dst, src, vtmp1);
2023 break;
2024 case 4:
2025 assert(vtmp2 == xnoreg, "");
2026 reduce4F(opcode, dst, src, vtmp1);
2027 break;
2028 case 8:
2029 reduce8F(opcode, dst, src, vtmp1, vtmp2);
2030 break;
2031 case 16:
2032 reduce16F(opcode, dst, src, vtmp1, vtmp2);
2033 break;
2034 default: assert(false, "wrong vector length");
2035 }
2036 }
2037
2038 void C2_MacroAssembler::reduceD(int opcode, int vlen, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2039 switch (vlen) {
2040 case 2:
2041 assert(vtmp2 == xnoreg, "");
2042 reduce2D(opcode, dst, src, vtmp1);
2043 break;
2044 case 4:
2045 reduce4D(opcode, dst, src, vtmp1, vtmp2);
2046 break;
2047 case 8:
2048 reduce8D(opcode, dst, src, vtmp1, vtmp2);
2049 break;
2050 default: assert(false, "wrong vector length");
2051 }
2052 }
2053
2054 void C2_MacroAssembler::unorderedReduceF(int opcode, int vlen, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2055 switch (vlen) {
2056 case 2:
2057 assert(vtmp1 == xnoreg, "");
2058 assert(vtmp2 == xnoreg, "");
2059 unorderedReduce2F(opcode, dst, src);
2060 break;
2061 case 4:
2062 assert(vtmp2 == xnoreg, "");
2063 unorderedReduce4F(opcode, dst, src, vtmp1);
2064 break;
2065 case 8:
2066 unorderedReduce8F(opcode, dst, src, vtmp1, vtmp2);
2067 break;
2068 case 16:
2069 unorderedReduce16F(opcode, dst, src, vtmp1, vtmp2);
2070 break;
2071 default: assert(false, "wrong vector length");
2072 }
2073 }
2074
2075 void C2_MacroAssembler::unorderedReduceD(int opcode, int vlen, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2076 switch (vlen) {
2077 case 2:
2078 assert(vtmp1 == xnoreg, "");
2079 assert(vtmp2 == xnoreg, "");
2080 unorderedReduce2D(opcode, dst, src);
2081 break;
2082 case 4:
2083 assert(vtmp2 == xnoreg, "");
2084 unorderedReduce4D(opcode, dst, src, vtmp1);
2085 break;
2086 case 8:
2087 unorderedReduce8D(opcode, dst, src, vtmp1, vtmp2);
2088 break;
2089 default: assert(false, "wrong vector length");
2090 }
2091 }
2092
2093 void C2_MacroAssembler::reduce2I(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2094 if (opcode == Op_AddReductionVI) {
2095 if (vtmp1 != src2) {
2096 movdqu(vtmp1, src2);
2097 }
2098 phaddd(vtmp1, vtmp1);
2099 } else {
2100 pshufd(vtmp1, src2, 0x1);
2101 reduce_operation_128(T_INT, opcode, vtmp1, src2);
2102 }
2103 movdl(vtmp2, src1);
2104 reduce_operation_128(T_INT, opcode, vtmp1, vtmp2);
2105 movdl(dst, vtmp1);
2106 }
2107
2108 void C2_MacroAssembler::reduce4I(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2109 if (opcode == Op_AddReductionVI) {
2110 if (vtmp1 != src2) {
2111 movdqu(vtmp1, src2);
2112 }
2113 phaddd(vtmp1, src2);
2114 reduce2I(opcode, dst, src1, vtmp1, vtmp1, vtmp2);
2115 } else {
2116 pshufd(vtmp2, src2, 0xE);
2117 reduce_operation_128(T_INT, opcode, vtmp2, src2);
2118 reduce2I(opcode, dst, src1, vtmp2, vtmp1, vtmp2);
2119 }
2120 }
2121
2122 void C2_MacroAssembler::reduce8I(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2123 if (opcode == Op_AddReductionVI) {
2124 vphaddd(vtmp1, src2, src2, Assembler::AVX_256bit);
2125 vextracti128_high(vtmp2, vtmp1);
2126 vpaddd(vtmp1, vtmp1, vtmp2, Assembler::AVX_128bit);
2127 reduce2I(opcode, dst, src1, vtmp1, vtmp1, vtmp2);
2128 } else {
2129 vextracti128_high(vtmp1, src2);
2130 reduce_operation_128(T_INT, opcode, vtmp1, src2);
2131 reduce4I(opcode, dst, src1, vtmp1, vtmp1, vtmp2);
2132 }
2133 }
2134
2135 void C2_MacroAssembler::reduce16I(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2136 vextracti64x4_high(vtmp2, src2);
2137 reduce_operation_256(T_INT, opcode, vtmp2, vtmp2, src2);
2138 reduce8I(opcode, dst, src1, vtmp2, vtmp1, vtmp2);
2139 }
2140
2141 void C2_MacroAssembler::reduce8B(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2142 pshufd(vtmp2, src2, 0x1);
2143 reduce_operation_128(T_BYTE, opcode, vtmp2, src2);
2144 movdqu(vtmp1, vtmp2);
2145 psrldq(vtmp1, 2);
2146 reduce_operation_128(T_BYTE, opcode, vtmp1, vtmp2);
2147 movdqu(vtmp2, vtmp1);
2148 psrldq(vtmp2, 1);
2149 reduce_operation_128(T_BYTE, opcode, vtmp1, vtmp2);
2150 movdl(vtmp2, src1);
2151 if (opcode == Op_UMinReductionV || opcode == Op_UMaxReductionV) {
2152 pmovzxbd(vtmp1, vtmp1);
2153 } else {
2154 pmovsxbd(vtmp1, vtmp1);
2155 }
2156 reduce_operation_128(T_INT, opcode, vtmp1, vtmp2);
2157 pextrb(dst, vtmp1, 0x0);
2158 movsbl(dst, dst);
2159 }
2160
2161 void C2_MacroAssembler::reduce16B(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2162 pshufd(vtmp1, src2, 0xE);
2163 reduce_operation_128(T_BYTE, opcode, vtmp1, src2);
2164 reduce8B(opcode, dst, src1, vtmp1, vtmp1, vtmp2);
2165 }
2166
2167 void C2_MacroAssembler::reduce32B(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2168 vextracti128_high(vtmp2, src2);
2169 reduce_operation_128(T_BYTE, opcode, vtmp2, src2);
2170 reduce16B(opcode, dst, src1, vtmp2, vtmp1, vtmp2);
2171 }
2172
2173 void C2_MacroAssembler::reduce64B(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2174 vextracti64x4_high(vtmp1, src2);
2175 reduce_operation_256(T_BYTE, opcode, vtmp1, vtmp1, src2);
2176 reduce32B(opcode, dst, src1, vtmp1, vtmp1, vtmp2);
2177 }
2178
2179 void C2_MacroAssembler::mulreduce8B(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2180 pmovsxbw(vtmp2, src2);
2181 reduce8S(opcode, dst, src1, vtmp2, vtmp1, vtmp2);
2182 }
2183
2184 void C2_MacroAssembler::mulreduce16B(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2185 if (UseAVX > 1) {
2186 int vector_len = Assembler::AVX_256bit;
2187 vpmovsxbw(vtmp1, src2, vector_len);
2188 reduce16S(opcode, dst, src1, vtmp1, vtmp1, vtmp2);
2189 } else {
2190 pmovsxbw(vtmp2, src2);
2191 reduce8S(opcode, dst, src1, vtmp2, vtmp1, vtmp2);
2192 pshufd(vtmp2, src2, 0xe);
2193 pmovsxbw(vtmp2, vtmp2);
2194 reduce8S(opcode, dst, dst, vtmp2, vtmp1, vtmp2);
2195 }
2196 }
2197
2198 void C2_MacroAssembler::mulreduce32B(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2199 if (UseAVX > 2 && VM_Version::supports_avx512bw()) {
2200 int vector_len = Assembler::AVX_512bit;
2201 vpmovsxbw(vtmp1, src2, vector_len);
2202 reduce32S(opcode, dst, src1, vtmp1, vtmp2, vtmp1);
2203 } else {
2204 assert(UseAVX >= 2,"Should not reach here.");
2205 mulreduce16B(opcode, dst, src1, src2, vtmp1, vtmp2);
2206 vextracti128_high(vtmp2, src2);
2207 mulreduce16B(opcode, dst, dst, vtmp2, vtmp1, vtmp2);
2208 }
2209 }
2210
2211 void C2_MacroAssembler::mulreduce64B(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2212 mulreduce32B(opcode, dst, src1, src2, vtmp1, vtmp2);
2213 vextracti64x4_high(vtmp2, src2);
2214 mulreduce32B(opcode, dst, dst, vtmp2, vtmp1, vtmp2);
2215 }
2216
2217 void C2_MacroAssembler::reduce4S(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2218 if (opcode == Op_AddReductionVI) {
2219 if (vtmp1 != src2) {
2220 movdqu(vtmp1, src2);
2221 }
2222 phaddw(vtmp1, vtmp1);
2223 phaddw(vtmp1, vtmp1);
2224 } else {
2225 pshufd(vtmp2, src2, 0x1);
2226 reduce_operation_128(T_SHORT, opcode, vtmp2, src2);
2227 movdqu(vtmp1, vtmp2);
2228 psrldq(vtmp1, 2);
2229 reduce_operation_128(T_SHORT, opcode, vtmp1, vtmp2);
2230 }
2231 movdl(vtmp2, src1);
2232 if (opcode == Op_UMinReductionV || opcode == Op_UMaxReductionV) {
2233 pmovzxwd(vtmp1, vtmp1);
2234 } else {
2235 pmovsxwd(vtmp1, vtmp1);
2236 }
2237 reduce_operation_128(T_INT, opcode, vtmp1, vtmp2);
2238 pextrw(dst, vtmp1, 0x0);
2239 movswl(dst, dst);
2240 }
2241
2242 void C2_MacroAssembler::reduce8S(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2243 if (opcode == Op_AddReductionVI) {
2244 if (vtmp1 != src2) {
2245 movdqu(vtmp1, src2);
2246 }
2247 phaddw(vtmp1, src2);
2248 } else {
2249 assert_different_registers(src2, vtmp1);
2250 pshufd(vtmp1, src2, 0xE);
2251 reduce_operation_128(T_SHORT, opcode, vtmp1, src2);
2252 }
2253 reduce4S(opcode, dst, src1, vtmp1, vtmp1, vtmp2);
2254 }
2255
2256 void C2_MacroAssembler::reduce16S(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2257 if (opcode == Op_AddReductionVI) {
2258 int vector_len = Assembler::AVX_256bit;
2259 vphaddw(vtmp2, src2, src2, vector_len);
2260 vpermq(vtmp2, vtmp2, 0xD8, vector_len);
2261 } else {
2262 assert_different_registers(src2, vtmp2);
2263 vextracti128_high(vtmp2, src2);
2264 reduce_operation_128(T_SHORT, opcode, vtmp2, src2);
2265 }
2266 reduce8S(opcode, dst, src1, vtmp2, vtmp1, vtmp2);
2267 }
2268
2269 void C2_MacroAssembler::reduce32S(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2270 assert_different_registers(src2, vtmp1);
2271 int vector_len = Assembler::AVX_256bit;
2272 vextracti64x4_high(vtmp1, src2);
2273 reduce_operation_256(T_SHORT, opcode, vtmp1, vtmp1, src2);
2274 reduce16S(opcode, dst, src1, vtmp1, vtmp1, vtmp2);
2275 }
2276
2277 void C2_MacroAssembler::reduce2L(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2278 pshufd(vtmp2, src2, 0xE);
2279 reduce_operation_128(T_LONG, opcode, vtmp2, src2);
2280 movdq(vtmp1, src1);
2281 reduce_operation_128(T_LONG, opcode, vtmp1, vtmp2);
2282 movdq(dst, vtmp1);
2283 }
2284
2285 void C2_MacroAssembler::reduce4L(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2286 vextracti128_high(vtmp1, src2);
2287 reduce_operation_128(T_LONG, opcode, vtmp1, src2);
2288 reduce2L(opcode, dst, src1, vtmp1, vtmp1, vtmp2);
2289 }
2290
2291 void C2_MacroAssembler::reduce8L(int opcode, Register dst, Register src1, XMMRegister src2, XMMRegister vtmp1, XMMRegister vtmp2) {
2292 vextracti64x4_high(vtmp2, src2);
2293 reduce_operation_256(T_LONG, opcode, vtmp2, vtmp2, src2);
2294 reduce4L(opcode, dst, src1, vtmp2, vtmp1, vtmp2);
2295 }
2296
2297 void C2_MacroAssembler::genmask(KRegister dst, Register len, Register temp) {
2298 mov64(temp, -1L);
2299 bzhiq(temp, temp, len);
2300 kmovql(dst, temp);
2301 }
2302
2303 void C2_MacroAssembler::reduce2F(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp) {
2304 reduce_operation_128(T_FLOAT, opcode, dst, src);
2305 pshufd(vtmp, src, 0x1);
2306 reduce_operation_128(T_FLOAT, opcode, dst, vtmp);
2307 }
2308
2309 void C2_MacroAssembler::reduce4F(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp) {
2310 reduce2F(opcode, dst, src, vtmp);
2311 pshufd(vtmp, src, 0x2);
2312 reduce_operation_128(T_FLOAT, opcode, dst, vtmp);
2313 pshufd(vtmp, src, 0x3);
2314 reduce_operation_128(T_FLOAT, opcode, dst, vtmp);
2315 }
2316
2317 void C2_MacroAssembler::reduce8F(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2318 reduce4F(opcode, dst, src, vtmp2);
2319 vextractf128_high(vtmp2, src);
2320 reduce4F(opcode, dst, vtmp2, vtmp1);
2321 }
2322
2323 void C2_MacroAssembler::reduce16F(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2324 reduce8F(opcode, dst, src, vtmp1, vtmp2);
2325 vextracti64x4_high(vtmp1, src);
2326 reduce8F(opcode, dst, vtmp1, vtmp1, vtmp2);
2327 }
2328
2329 void C2_MacroAssembler::unorderedReduce2F(int opcode, XMMRegister dst, XMMRegister src) {
2330 pshufd(dst, src, 0x1);
2331 reduce_operation_128(T_FLOAT, opcode, dst, src);
2332 }
2333
2334 void C2_MacroAssembler::unorderedReduce4F(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp) {
2335 pshufd(vtmp, src, 0xE);
2336 unordered_reduce_operation_128(T_FLOAT, opcode, vtmp, src);
2337 unorderedReduce2F(opcode, dst, vtmp);
2338 }
2339
2340 void C2_MacroAssembler::unorderedReduce8F(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2341 vextractf128_high(vtmp1, src);
2342 unordered_reduce_operation_128(T_FLOAT, opcode, vtmp1, src);
2343 unorderedReduce4F(opcode, dst, vtmp1, vtmp2);
2344 }
2345
2346 void C2_MacroAssembler::unorderedReduce16F(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2347 vextractf64x4_high(vtmp2, src);
2348 unordered_reduce_operation_256(T_FLOAT, opcode, vtmp2, vtmp2, src);
2349 unorderedReduce8F(opcode, dst, vtmp2, vtmp1, vtmp2);
2350 }
2351
2352 void C2_MacroAssembler::reduce2D(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp) {
2353 reduce_operation_128(T_DOUBLE, opcode, dst, src);
2354 pshufd(vtmp, src, 0xE);
2355 reduce_operation_128(T_DOUBLE, opcode, dst, vtmp);
2356 }
2357
2358 void C2_MacroAssembler::reduce4D(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2359 reduce2D(opcode, dst, src, vtmp2);
2360 vextractf128_high(vtmp2, src);
2361 reduce2D(opcode, dst, vtmp2, vtmp1);
2362 }
2363
2364 void C2_MacroAssembler::reduce8D(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2365 reduce4D(opcode, dst, src, vtmp1, vtmp2);
2366 vextracti64x4_high(vtmp1, src);
2367 reduce4D(opcode, dst, vtmp1, vtmp1, vtmp2);
2368 }
2369
2370 void C2_MacroAssembler::unorderedReduce2D(int opcode, XMMRegister dst, XMMRegister src) {
2371 pshufd(dst, src, 0xE);
2372 reduce_operation_128(T_DOUBLE, opcode, dst, src);
2373 }
2374
2375 void C2_MacroAssembler::unorderedReduce4D(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp) {
2376 vextractf128_high(vtmp, src);
2377 unordered_reduce_operation_128(T_DOUBLE, opcode, vtmp, src);
2378 unorderedReduce2D(opcode, dst, vtmp);
2379 }
2380
2381 void C2_MacroAssembler::unorderedReduce8D(int opcode, XMMRegister dst, XMMRegister src, XMMRegister vtmp1, XMMRegister vtmp2) {
2382 vextractf64x4_high(vtmp2, src);
2383 unordered_reduce_operation_256(T_DOUBLE, opcode, vtmp2, vtmp2, src);
2384 unorderedReduce4D(opcode, dst, vtmp2, vtmp1);
2385 }
2386
2387 void C2_MacroAssembler::evmovdqu(BasicType type, KRegister kmask, XMMRegister dst, Address src, bool merge, int vector_len) {
2388 MacroAssembler::evmovdqu(type, kmask, dst, src, merge, vector_len);
2389 }
2390
2391 void C2_MacroAssembler::evmovdqu(BasicType type, KRegister kmask, Address dst, XMMRegister src, bool merge, int vector_len) {
2392 MacroAssembler::evmovdqu(type, kmask, dst, src, merge, vector_len);
2393 }
2394
2395 void C2_MacroAssembler::evmovdqu(BasicType type, KRegister kmask, XMMRegister dst, XMMRegister src, bool merge, int vector_len) {
2396 MacroAssembler::evmovdqu(type, kmask, dst, src, merge, vector_len);
2397 }
2398
2399 void C2_MacroAssembler::vmovmask(BasicType elem_bt, XMMRegister dst, Address src, XMMRegister mask,
2400 int vec_enc) {
2401 switch(elem_bt) {
2402 case T_INT:
2403 case T_FLOAT:
2404 vmaskmovps(dst, src, mask, vec_enc);
2405 break;
2406 case T_LONG:
2407 case T_DOUBLE:
2408 vmaskmovpd(dst, src, mask, vec_enc);
2409 break;
2410 default:
2411 fatal("Unsupported type %s", type2name(elem_bt));
2412 break;
2413 }
2414 }
2415
2416 void C2_MacroAssembler::vmovmask(BasicType elem_bt, Address dst, XMMRegister src, XMMRegister mask,
2417 int vec_enc) {
2418 switch(elem_bt) {
2419 case T_INT:
2420 case T_FLOAT:
2421 vmaskmovps(dst, src, mask, vec_enc);
2422 break;
2423 case T_LONG:
2424 case T_DOUBLE:
2425 vmaskmovpd(dst, src, mask, vec_enc);
2426 break;
2427 default:
2428 fatal("Unsupported type %s", type2name(elem_bt));
2429 break;
2430 }
2431 }
2432
2433 void C2_MacroAssembler::reduceFloatMinMax(int opcode, int vlen, bool is_dst_valid,
2434 XMMRegister dst, XMMRegister src,
2435 XMMRegister tmp, XMMRegister atmp, XMMRegister btmp,
2436 XMMRegister xmm_0, XMMRegister xmm_1) {
2437 const int permconst[] = {1, 14};
2438 XMMRegister wsrc = src;
2439 XMMRegister wdst = xmm_0;
2440 XMMRegister wtmp = (xmm_1 == xnoreg) ? xmm_0: xmm_1;
2441
2442 int vlen_enc = Assembler::AVX_128bit;
2443 if (vlen == 16) {
2444 vlen_enc = Assembler::AVX_256bit;
2445 }
2446
2447 for (int i = log2(vlen) - 1; i >=0; i--) {
2448 if (i == 0 && !is_dst_valid) {
2449 wdst = dst;
2450 }
2451 if (i == 3) {
2452 vextracti64x4_high(wtmp, wsrc);
2453 } else if (i == 2) {
2454 vextracti128_high(wtmp, wsrc);
2455 } else { // i = [0,1]
2456 vpermilps(wtmp, wsrc, permconst[i], vlen_enc);
2457 }
2458
2459 if (VM_Version::supports_avx10_2()) {
2460 vminmax_fp_avx10_2(opcode, T_FLOAT, wdst, k0, wtmp, wsrc, vlen_enc);
2461 } else {
2462 vminmax_fp(opcode, T_FLOAT, wdst, wtmp, wsrc, tmp, atmp, btmp, vlen_enc);
2463 }
2464 wsrc = wdst;
2465 vlen_enc = Assembler::AVX_128bit;
2466 }
2467 if (is_dst_valid) {
2468 if (VM_Version::supports_avx10_2()) {
2469 vminmax_fp_avx10_2(opcode, T_FLOAT, dst, k0, wdst, dst, Assembler::AVX_128bit);
2470 } else {
2471 vminmax_fp(opcode, T_FLOAT, dst, wdst, dst, tmp, atmp, btmp, Assembler::AVX_128bit);
2472 }
2473 }
2474 }
2475
2476 void C2_MacroAssembler::reduceDoubleMinMax(int opcode, int vlen, bool is_dst_valid, XMMRegister dst, XMMRegister src,
2477 XMMRegister tmp, XMMRegister atmp, XMMRegister btmp,
2478 XMMRegister xmm_0, XMMRegister xmm_1) {
2479 XMMRegister wsrc = src;
2480 XMMRegister wdst = xmm_0;
2481 XMMRegister wtmp = (xmm_1 == xnoreg) ? xmm_0: xmm_1;
2482 int vlen_enc = Assembler::AVX_128bit;
2483 if (vlen == 8) {
2484 vlen_enc = Assembler::AVX_256bit;
2485 }
2486 for (int i = log2(vlen) - 1; i >=0; i--) {
2487 if (i == 0 && !is_dst_valid) {
2488 wdst = dst;
2489 }
2490 if (i == 1) {
2491 vextracti128_high(wtmp, wsrc);
2492 } else if (i == 2) {
2493 vextracti64x4_high(wtmp, wsrc);
2494 } else {
2495 assert(i == 0, "%d", i);
2496 vpermilpd(wtmp, wsrc, 1, vlen_enc);
2497 }
2498
2499 if (VM_Version::supports_avx10_2()) {
2500 vminmax_fp_avx10_2(opcode, T_DOUBLE, wdst, k0, wtmp, wsrc, vlen_enc);
2501 } else {
2502 vminmax_fp(opcode, T_DOUBLE, wdst, wtmp, wsrc, tmp, atmp, btmp, vlen_enc);
2503 }
2504
2505 wsrc = wdst;
2506 vlen_enc = Assembler::AVX_128bit;
2507 }
2508
2509 if (is_dst_valid) {
2510 if (VM_Version::supports_avx10_2()) {
2511 vminmax_fp_avx10_2(opcode, T_DOUBLE, dst, k0, wdst, dst, Assembler::AVX_128bit);
2512 } else {
2513 vminmax_fp(opcode, T_DOUBLE, dst, wdst, dst, tmp, atmp, btmp, Assembler::AVX_128bit);
2514 }
2515 }
2516 }
2517
2518 void C2_MacroAssembler::extract(BasicType bt, Register dst, XMMRegister src, int idx) {
2519 switch (bt) {
2520 case T_BYTE: pextrb(dst, src, idx); break;
2521 case T_SHORT: pextrw(dst, src, idx); break;
2522 case T_INT: pextrd(dst, src, idx); break;
2523 case T_LONG: pextrq(dst, src, idx); break;
2524
2525 default:
2526 assert(false,"Should not reach here.");
2527 break;
2528 }
2529 }
2530
2531 XMMRegister C2_MacroAssembler::get_lane(BasicType typ, XMMRegister dst, XMMRegister src, int elemindex) {
2532 int esize = type2aelembytes(typ);
2533 int elem_per_lane = 16/esize;
2534 int lane = elemindex / elem_per_lane;
2535 int eindex = elemindex % elem_per_lane;
2536
2537 if (lane >= 2) {
2538 assert(UseAVX > 2, "required");
2539 vextractf32x4(dst, src, lane & 3);
2540 return dst;
2541 } else if (lane > 0) {
2542 assert(UseAVX > 0, "required");
2543 vextractf128(dst, src, lane);
2544 return dst;
2545 } else {
2546 return src;
2547 }
2548 }
2549
2550 void C2_MacroAssembler::movsxl(BasicType typ, Register dst) {
2551 if (typ == T_BYTE) {
2552 movsbl(dst, dst);
2553 } else if (typ == T_SHORT) {
2554 movswl(dst, dst);
2555 }
2556 }
2557
2558 void C2_MacroAssembler::get_elem(BasicType typ, Register dst, XMMRegister src, int elemindex) {
2559 int esize = type2aelembytes(typ);
2560 int elem_per_lane = 16/esize;
2561 int eindex = elemindex % elem_per_lane;
2562 assert(is_integral_type(typ),"required");
2563
2564 if (eindex == 0) {
2565 if (typ == T_LONG) {
2566 movq(dst, src);
2567 } else {
2568 movdl(dst, src);
2569 movsxl(typ, dst);
2570 }
2571 } else {
2572 extract(typ, dst, src, eindex);
2573 movsxl(typ, dst);
2574 }
2575 }
2576
2577 void C2_MacroAssembler::get_elem(BasicType typ, XMMRegister dst, XMMRegister src, int elemindex, XMMRegister vtmp) {
2578 int esize = type2aelembytes(typ);
2579 int elem_per_lane = 16/esize;
2580 int eindex = elemindex % elem_per_lane;
2581 assert((typ == T_FLOAT || typ == T_DOUBLE),"required");
2582
2583 if (eindex == 0) {
2584 movq(dst, src);
2585 } else {
2586 if (typ == T_FLOAT) {
2587 if (UseAVX == 0) {
2588 movdqu(dst, src);
2589 shufps(dst, dst, eindex);
2590 } else {
2591 vshufps(dst, src, src, eindex, Assembler::AVX_128bit);
2592 }
2593 } else {
2594 if (UseAVX == 0) {
2595 movdqu(dst, src);
2596 psrldq(dst, eindex*esize);
2597 } else {
2598 vpsrldq(dst, src, eindex*esize, Assembler::AVX_128bit);
2599 }
2600 movq(dst, dst);
2601 }
2602 }
2603 // Zero upper bits
2604 if (typ == T_FLOAT) {
2605 if (UseAVX == 0) {
2606 assert(vtmp != xnoreg, "required.");
2607 movdqu(vtmp, ExternalAddress(StubRoutines::x86::vector_32_bit_mask()), noreg);
2608 pand(dst, vtmp);
2609 } else {
2610 vpand(dst, dst, ExternalAddress(StubRoutines::x86::vector_32_bit_mask()), Assembler::AVX_128bit, noreg);
2611 }
2612 }
2613 }
2614
2615 void C2_MacroAssembler::evpcmp(BasicType typ, KRegister kdmask, KRegister ksmask, XMMRegister src1, XMMRegister src2, int comparison, int vector_len) {
2616 switch(typ) {
2617 case T_BYTE:
2618 case T_BOOLEAN:
2619 evpcmpb(kdmask, ksmask, src1, src2, comparison, /*signed*/ true, vector_len);
2620 break;
2621 case T_SHORT:
2622 case T_CHAR:
2623 evpcmpw(kdmask, ksmask, src1, src2, comparison, /*signed*/ true, vector_len);
2624 break;
2625 case T_INT:
2626 case T_FLOAT:
2627 evpcmpd(kdmask, ksmask, src1, src2, comparison, /*signed*/ true, vector_len);
2628 break;
2629 case T_LONG:
2630 case T_DOUBLE:
2631 evpcmpq(kdmask, ksmask, src1, src2, comparison, /*signed*/ true, vector_len);
2632 break;
2633 default:
2634 assert(false,"Should not reach here.");
2635 break;
2636 }
2637 }
2638
2639 void C2_MacroAssembler::evpcmp(BasicType typ, KRegister kdmask, KRegister ksmask, XMMRegister src1, AddressLiteral src2, int comparison, int vector_len, Register rscratch) {
2640 assert(rscratch != noreg || always_reachable(src2), "missing");
2641
2642 switch(typ) {
2643 case T_BOOLEAN:
2644 case T_BYTE:
2645 evpcmpb(kdmask, ksmask, src1, src2, comparison, /*signed*/ true, vector_len, rscratch);
2646 break;
2647 case T_CHAR:
2648 case T_SHORT:
2649 evpcmpw(kdmask, ksmask, src1, src2, comparison, /*signed*/ true, vector_len, rscratch);
2650 break;
2651 case T_INT:
2652 case T_FLOAT:
2653 evpcmpd(kdmask, ksmask, src1, src2, comparison, /*signed*/ true, vector_len, rscratch);
2654 break;
2655 case T_LONG:
2656 case T_DOUBLE:
2657 evpcmpq(kdmask, ksmask, src1, src2, comparison, /*signed*/ true, vector_len, rscratch);
2658 break;
2659 default:
2660 assert(false,"Should not reach here.");
2661 break;
2662 }
2663 }
2664
2665 void C2_MacroAssembler::evpblend(BasicType typ, XMMRegister dst, KRegister kmask, XMMRegister src1, XMMRegister src2, bool merge, int vector_len) {
2666 switch(typ) {
2667 case T_BYTE:
2668 evpblendmb(dst, kmask, src1, src2, merge, vector_len);
2669 break;
2670 case T_SHORT:
2671 evpblendmw(dst, kmask, src1, src2, merge, vector_len);
2672 break;
2673 case T_INT:
2674 case T_FLOAT:
2675 evpblendmd(dst, kmask, src1, src2, merge, vector_len);
2676 break;
2677 case T_LONG:
2678 case T_DOUBLE:
2679 evpblendmq(dst, kmask, src1, src2, merge, vector_len);
2680 break;
2681 default:
2682 assert(false,"Should not reach here.");
2683 break;
2684 }
2685 }
2686
2687 void C2_MacroAssembler::vectortest(BasicType bt, XMMRegister src1, XMMRegister src2, XMMRegister vtmp, int vlen_in_bytes) {
2688 assert(vlen_in_bytes <= 32, "");
2689 int esize = type2aelembytes(bt);
2690 if (vlen_in_bytes == 32) {
2691 assert(vtmp == xnoreg, "required.");
2692 if (esize >= 4) {
2693 vtestps(src1, src2, AVX_256bit);
2694 } else {
2695 vptest(src1, src2, AVX_256bit);
2696 }
2697 return;
2698 }
2699 if (vlen_in_bytes < 16) {
2700 // Duplicate the lower part to fill the whole register,
2701 // Don't need to do so for src2
2702 assert(vtmp != xnoreg, "required");
2703 int shuffle_imm = (vlen_in_bytes == 4) ? 0x00 : 0x04;
2704 pshufd(vtmp, src1, shuffle_imm);
2705 } else {
2706 assert(vtmp == xnoreg, "required");
2707 vtmp = src1;
2708 }
2709 if (esize >= 4 && VM_Version::supports_avx()) {
2710 vtestps(vtmp, src2, AVX_128bit);
2711 } else {
2712 ptest(vtmp, src2);
2713 }
2714 }
2715
2716 void C2_MacroAssembler::vpadd(BasicType elem_bt, XMMRegister dst, XMMRegister src1, XMMRegister src2, int vlen_enc) {
2717 #ifdef ASSERT
2718 bool is_bw = ((elem_bt == T_BYTE) || (elem_bt == T_SHORT));
2719 bool is_bw_supported = VM_Version::supports_avx512bw();
2720 if (is_bw && !is_bw_supported) {
2721 assert(vlen_enc != Assembler::AVX_512bit, "required");
2722 assert((dst->encoding() < 16) && (src1->encoding() < 16) && (src2->encoding() < 16),
2723 "XMM register should be 0-15");
2724 }
2725 #endif // ASSERT
2726 switch (elem_bt) {
2727 case T_BYTE: vpaddb(dst, src1, src2, vlen_enc); return;
2728 case T_SHORT: vpaddw(dst, src1, src2, vlen_enc); return;
2729 case T_INT: vpaddd(dst, src1, src2, vlen_enc); return;
2730 case T_FLOAT: vaddps(dst, src1, src2, vlen_enc); return;
2731 case T_LONG: vpaddq(dst, src1, src2, vlen_enc); return;
2732 case T_DOUBLE: vaddpd(dst, src1, src2, vlen_enc); return;
2733 default: fatal("Unsupported type %s", type2name(elem_bt)); return;
2734 }
2735 }
2736
2737 void C2_MacroAssembler::vpbroadcast(BasicType elem_bt, XMMRegister dst, Register src, int vlen_enc) {
2738 assert(UseAVX >= 2, "required");
2739 bool is_bw = ((elem_bt == T_BYTE) || (elem_bt == T_SHORT));
2740 bool is_vl = vlen_enc != Assembler::AVX_512bit;
2741 if ((UseAVX > 2) &&
2742 (!is_bw || VM_Version::supports_avx512bw()) &&
2743 (!is_vl || VM_Version::supports_avx512vl())) {
2744 switch (elem_bt) {
2745 case T_BYTE: evpbroadcastb(dst, src, vlen_enc); return;
2746 case T_SHORT: evpbroadcastw(dst, src, vlen_enc); return;
2747 case T_FLOAT: case T_INT: evpbroadcastd(dst, src, vlen_enc); return;
2748 case T_DOUBLE: case T_LONG: evpbroadcastq(dst, src, vlen_enc); return;
2749 default: fatal("Unsupported type %s", type2name(elem_bt)); return;
2750 }
2751 } else {
2752 assert(vlen_enc != Assembler::AVX_512bit, "required");
2753 assert((dst->encoding() < 16),"XMM register should be 0-15");
2754 switch (elem_bt) {
2755 case T_BYTE: movdl(dst, src); vpbroadcastb(dst, dst, vlen_enc); return;
2756 case T_SHORT: movdl(dst, src); vpbroadcastw(dst, dst, vlen_enc); return;
2757 case T_INT: movdl(dst, src); vpbroadcastd(dst, dst, vlen_enc); return;
2758 case T_FLOAT: movdl(dst, src); vbroadcastss(dst, dst, vlen_enc); return;
2759 case T_LONG: movdq(dst, src); vpbroadcastq(dst, dst, vlen_enc); return;
2760 case T_DOUBLE: movdq(dst, src); vbroadcastsd(dst, dst, vlen_enc); return;
2761 default: fatal("Unsupported type %s", type2name(elem_bt)); return;
2762 }
2763 }
2764 }
2765
2766 void C2_MacroAssembler::vconvert_b2x(BasicType to_elem_bt, XMMRegister dst, XMMRegister src, int vlen_enc) {
2767 switch (to_elem_bt) {
2768 case T_SHORT:
2769 vpmovsxbw(dst, src, vlen_enc);
2770 break;
2771 case T_INT:
2772 vpmovsxbd(dst, src, vlen_enc);
2773 break;
2774 case T_FLOAT:
2775 vpmovsxbd(dst, src, vlen_enc);
2776 vcvtdq2ps(dst, dst, vlen_enc);
2777 break;
2778 case T_LONG:
2779 vpmovsxbq(dst, src, vlen_enc);
2780 break;
2781 case T_DOUBLE: {
2782 int mid_vlen_enc = (vlen_enc == Assembler::AVX_512bit) ? Assembler::AVX_256bit : Assembler::AVX_128bit;
2783 vpmovsxbd(dst, src, mid_vlen_enc);
2784 vcvtdq2pd(dst, dst, vlen_enc);
2785 break;
2786 }
2787 default:
2788 fatal("Unsupported type %s", type2name(to_elem_bt));
2789 break;
2790 }
2791 }
2792
2793 //-------------------------------------------------------------------------------------------
2794
2795 // IndexOf for constant substrings with size >= 8 chars
2796 // which don't need to be loaded through stack.
2797 void C2_MacroAssembler::string_indexofC8(Register str1, Register str2,
2798 Register cnt1, Register cnt2,
2799 int int_cnt2, Register result,
2800 XMMRegister vec, Register tmp,
2801 int ae) {
2802 ShortBranchVerifier sbv(this);
2803 assert(UseSSE42Intrinsics, "SSE4.2 intrinsics are required");
2804 assert(ae != StrIntrinsicNode::LU, "Invalid encoding");
2805
2806 // This method uses the pcmpestri instruction with bound registers
2807 // inputs:
2808 // xmm - substring
2809 // rax - substring length (elements count)
2810 // mem - scanned string
2811 // rdx - string length (elements count)
2812 // 0xd - mode: 1100 (substring search) + 01 (unsigned shorts)
2813 // 0xc - mode: 1100 (substring search) + 00 (unsigned bytes)
2814 // outputs:
2815 // rcx - matched index in string
2816 assert(cnt1 == rdx && cnt2 == rax && tmp == rcx, "pcmpestri");
2817 int mode = (ae == StrIntrinsicNode::LL) ? 0x0c : 0x0d; // bytes or shorts
2818 int stride = (ae == StrIntrinsicNode::LL) ? 16 : 8; //UU, UL -> 8
2819 Address::ScaleFactor scale1 = (ae == StrIntrinsicNode::LL) ? Address::times_1 : Address::times_2;
2820 Address::ScaleFactor scale2 = (ae == StrIntrinsicNode::UL) ? Address::times_1 : scale1;
2821
2822 Label RELOAD_SUBSTR, SCAN_TO_SUBSTR, SCAN_SUBSTR,
2823 RET_FOUND, RET_NOT_FOUND, EXIT, FOUND_SUBSTR,
2824 MATCH_SUBSTR_HEAD, RELOAD_STR, FOUND_CANDIDATE;
2825
2826 // Note, inline_string_indexOf() generates checks:
2827 // if (substr.count > string.count) return -1;
2828 // if (substr.count == 0) return 0;
2829 assert(int_cnt2 >= stride, "this code is used only for cnt2 >= 8 chars");
2830
2831 // Load substring.
2832 if (ae == StrIntrinsicNode::UL) {
2833 pmovzxbw(vec, Address(str2, 0));
2834 } else {
2835 movdqu(vec, Address(str2, 0));
2836 }
2837 movl(cnt2, int_cnt2);
2838 movptr(result, str1); // string addr
2839
2840 if (int_cnt2 > stride) {
2841 jmpb(SCAN_TO_SUBSTR);
2842
2843 // Reload substr for rescan, this code
2844 // is executed only for large substrings (> 8 chars)
2845 bind(RELOAD_SUBSTR);
2846 if (ae == StrIntrinsicNode::UL) {
2847 pmovzxbw(vec, Address(str2, 0));
2848 } else {
2849 movdqu(vec, Address(str2, 0));
2850 }
2851 negptr(cnt2); // Jumped here with negative cnt2, convert to positive
2852
2853 bind(RELOAD_STR);
2854 // We came here after the beginning of the substring was
2855 // matched but the rest of it was not so we need to search
2856 // again. Start from the next element after the previous match.
2857
2858 // cnt2 is number of substring reminding elements and
2859 // cnt1 is number of string reminding elements when cmp failed.
2860 // Restored cnt1 = cnt1 - cnt2 + int_cnt2
2861 subl(cnt1, cnt2);
2862 addl(cnt1, int_cnt2);
2863 movl(cnt2, int_cnt2); // Now restore cnt2
2864
2865 decrementl(cnt1); // Shift to next element
2866 cmpl(cnt1, cnt2);
2867 jcc(Assembler::negative, RET_NOT_FOUND); // Left less then substring
2868
2869 addptr(result, (1<<scale1));
2870
2871 } // (int_cnt2 > 8)
2872
2873 // Scan string for start of substr in 16-byte vectors
2874 bind(SCAN_TO_SUBSTR);
2875 pcmpestri(vec, Address(result, 0), mode);
2876 jccb(Assembler::below, FOUND_CANDIDATE); // CF == 1
2877 subl(cnt1, stride);
2878 jccb(Assembler::lessEqual, RET_NOT_FOUND); // Scanned full string
2879 cmpl(cnt1, cnt2);
2880 jccb(Assembler::negative, RET_NOT_FOUND); // Left less then substring
2881 addptr(result, 16);
2882 jmpb(SCAN_TO_SUBSTR);
2883
2884 // Found a potential substr
2885 bind(FOUND_CANDIDATE);
2886 // Matched whole vector if first element matched (tmp(rcx) == 0).
2887 if (int_cnt2 == stride) {
2888 jccb(Assembler::overflow, RET_FOUND); // OF == 1
2889 } else { // int_cnt2 > 8
2890 jccb(Assembler::overflow, FOUND_SUBSTR);
2891 }
2892 // After pcmpestri tmp(rcx) contains matched element index
2893 // Compute start addr of substr
2894 lea(result, Address(result, tmp, scale1));
2895
2896 // Make sure string is still long enough
2897 subl(cnt1, tmp);
2898 cmpl(cnt1, cnt2);
2899 if (int_cnt2 == stride) {
2900 jccb(Assembler::greaterEqual, SCAN_TO_SUBSTR);
2901 } else { // int_cnt2 > 8
2902 jccb(Assembler::greaterEqual, MATCH_SUBSTR_HEAD);
2903 }
2904 // Left less then substring.
2905
2906 bind(RET_NOT_FOUND);
2907 movl(result, -1);
2908 jmp(EXIT);
2909
2910 if (int_cnt2 > stride) {
2911 // This code is optimized for the case when whole substring
2912 // is matched if its head is matched.
2913 bind(MATCH_SUBSTR_HEAD);
2914 pcmpestri(vec, Address(result, 0), mode);
2915 // Reload only string if does not match
2916 jcc(Assembler::noOverflow, RELOAD_STR); // OF == 0
2917
2918 Label CONT_SCAN_SUBSTR;
2919 // Compare the rest of substring (> 8 chars).
2920 bind(FOUND_SUBSTR);
2921 // First 8 chars are already matched.
2922 negptr(cnt2);
2923 addptr(cnt2, stride);
2924
2925 bind(SCAN_SUBSTR);
2926 subl(cnt1, stride);
2927 cmpl(cnt2, -stride); // Do not read beyond substring
2928 jccb(Assembler::lessEqual, CONT_SCAN_SUBSTR);
2929 // Back-up strings to avoid reading beyond substring:
2930 // cnt1 = cnt1 - cnt2 + 8
2931 addl(cnt1, cnt2); // cnt2 is negative
2932 addl(cnt1, stride);
2933 movl(cnt2, stride); negptr(cnt2);
2934 bind(CONT_SCAN_SUBSTR);
2935 if (int_cnt2 < (int)G) {
2936 int tail_off1 = int_cnt2<<scale1;
2937 int tail_off2 = int_cnt2<<scale2;
2938 if (ae == StrIntrinsicNode::UL) {
2939 pmovzxbw(vec, Address(str2, cnt2, scale2, tail_off2));
2940 } else {
2941 movdqu(vec, Address(str2, cnt2, scale2, tail_off2));
2942 }
2943 pcmpestri(vec, Address(result, cnt2, scale1, tail_off1), mode);
2944 } else {
2945 // calculate index in register to avoid integer overflow (int_cnt2*2)
2946 movl(tmp, int_cnt2);
2947 addptr(tmp, cnt2);
2948 if (ae == StrIntrinsicNode::UL) {
2949 pmovzxbw(vec, Address(str2, tmp, scale2, 0));
2950 } else {
2951 movdqu(vec, Address(str2, tmp, scale2, 0));
2952 }
2953 pcmpestri(vec, Address(result, tmp, scale1, 0), mode);
2954 }
2955 // Need to reload strings pointers if not matched whole vector
2956 jcc(Assembler::noOverflow, RELOAD_SUBSTR); // OF == 0
2957 addptr(cnt2, stride);
2958 jcc(Assembler::negative, SCAN_SUBSTR);
2959 // Fall through if found full substring
2960
2961 } // (int_cnt2 > 8)
2962
2963 bind(RET_FOUND);
2964 // Found result if we matched full small substring.
2965 // Compute substr offset
2966 subptr(result, str1);
2967 if (ae == StrIntrinsicNode::UU || ae == StrIntrinsicNode::UL) {
2968 shrl(result, 1); // index
2969 }
2970 bind(EXIT);
2971
2972 } // string_indexofC8
2973
2974 // Small strings are loaded through stack if they cross page boundary.
2975 void C2_MacroAssembler::string_indexof(Register str1, Register str2,
2976 Register cnt1, Register cnt2,
2977 int int_cnt2, Register result,
2978 XMMRegister vec, Register tmp,
2979 int ae) {
2980 ShortBranchVerifier sbv(this);
2981 assert(UseSSE42Intrinsics, "SSE4.2 intrinsics are required");
2982 assert(ae != StrIntrinsicNode::LU, "Invalid encoding");
2983
2984 //
2985 // int_cnt2 is length of small (< 8 chars) constant substring
2986 // or (-1) for non constant substring in which case its length
2987 // is in cnt2 register.
2988 //
2989 // Note, inline_string_indexOf() generates checks:
2990 // if (substr.count > string.count) return -1;
2991 // if (substr.count == 0) return 0;
2992 //
2993 int stride = (ae == StrIntrinsicNode::LL) ? 16 : 8; //UU, UL -> 8
2994 assert(int_cnt2 == -1 || (0 < int_cnt2 && int_cnt2 < stride), "should be != 0");
2995 // This method uses the pcmpestri instruction with bound registers
2996 // inputs:
2997 // xmm - substring
2998 // rax - substring length (elements count)
2999 // mem - scanned string
3000 // rdx - string length (elements count)
3001 // 0xd - mode: 1100 (substring search) + 01 (unsigned shorts)
3002 // 0xc - mode: 1100 (substring search) + 00 (unsigned bytes)
3003 // outputs:
3004 // rcx - matched index in string
3005 assert(cnt1 == rdx && cnt2 == rax && tmp == rcx, "pcmpestri");
3006 int mode = (ae == StrIntrinsicNode::LL) ? 0x0c : 0x0d; // bytes or shorts
3007 Address::ScaleFactor scale1 = (ae == StrIntrinsicNode::LL) ? Address::times_1 : Address::times_2;
3008 Address::ScaleFactor scale2 = (ae == StrIntrinsicNode::UL) ? Address::times_1 : scale1;
3009
3010 Label RELOAD_SUBSTR, SCAN_TO_SUBSTR, SCAN_SUBSTR, ADJUST_STR,
3011 RET_FOUND, RET_NOT_FOUND, CLEANUP, FOUND_SUBSTR,
3012 FOUND_CANDIDATE;
3013
3014 { //========================================================
3015 // We don't know where these strings are located
3016 // and we can't read beyond them. Load them through stack.
3017 Label BIG_STRINGS, CHECK_STR, COPY_SUBSTR, COPY_STR;
3018
3019 movptr(tmp, rsp); // save old SP
3020
3021 if (int_cnt2 > 0) { // small (< 8 chars) constant substring
3022 if (int_cnt2 == (1>>scale2)) { // One byte
3023 assert((ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UL), "Only possible for latin1 encoding");
3024 load_unsigned_byte(result, Address(str2, 0));
3025 movdl(vec, result); // move 32 bits
3026 } else if (ae == StrIntrinsicNode::LL && int_cnt2 == 3) { // Three bytes
3027 // Not enough header space in 32-bit VM: 12+3 = 15.
3028 movl(result, Address(str2, -1));
3029 shrl(result, 8);
3030 movdl(vec, result); // move 32 bits
3031 } else if (ae != StrIntrinsicNode::UL && int_cnt2 == (2>>scale2)) { // One char
3032 load_unsigned_short(result, Address(str2, 0));
3033 movdl(vec, result); // move 32 bits
3034 } else if (ae != StrIntrinsicNode::UL && int_cnt2 == (4>>scale2)) { // Two chars
3035 movdl(vec, Address(str2, 0)); // move 32 bits
3036 } else if (ae != StrIntrinsicNode::UL && int_cnt2 == (8>>scale2)) { // Four chars
3037 movq(vec, Address(str2, 0)); // move 64 bits
3038 } else { // cnt2 = { 3, 5, 6, 7 } || (ae == StrIntrinsicNode::UL && cnt2 ={2, ..., 7})
3039 // Array header size is 12 bytes in 32-bit VM
3040 // + 6 bytes for 3 chars == 18 bytes,
3041 // enough space to load vec and shift.
3042 assert(HeapWordSize*TypeArrayKlass::header_size() >= 12,"sanity");
3043 if (ae == StrIntrinsicNode::UL) {
3044 int tail_off = int_cnt2-8;
3045 pmovzxbw(vec, Address(str2, tail_off));
3046 psrldq(vec, -2*tail_off);
3047 }
3048 else {
3049 int tail_off = int_cnt2*(1<<scale2);
3050 movdqu(vec, Address(str2, tail_off-16));
3051 psrldq(vec, 16-tail_off);
3052 }
3053 }
3054 } else { // not constant substring
3055 cmpl(cnt2, stride);
3056 jccb(Assembler::aboveEqual, BIG_STRINGS); // Both strings are big enough
3057
3058 // We can read beyond string if srt+16 does not cross page boundary
3059 // since heaps are aligned and mapped by pages.
3060 assert(os::vm_page_size() < (int)G, "default page should be small");
3061 movl(result, str2); // We need only low 32 bits
3062 andl(result, ((int)os::vm_page_size()-1));
3063 cmpl(result, ((int)os::vm_page_size()-16));
3064 jccb(Assembler::belowEqual, CHECK_STR);
3065
3066 // Move small strings to stack to allow load 16 bytes into vec.
3067 subptr(rsp, 16);
3068 int stk_offset = wordSize-(1<<scale2);
3069 push(cnt2);
3070
3071 bind(COPY_SUBSTR);
3072 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UL) {
3073 load_unsigned_byte(result, Address(str2, cnt2, scale2, -1));
3074 movb(Address(rsp, cnt2, scale2, stk_offset), result);
3075 } else if (ae == StrIntrinsicNode::UU) {
3076 load_unsigned_short(result, Address(str2, cnt2, scale2, -2));
3077 movw(Address(rsp, cnt2, scale2, stk_offset), result);
3078 }
3079 decrement(cnt2);
3080 jccb(Assembler::notZero, COPY_SUBSTR);
3081
3082 pop(cnt2);
3083 movptr(str2, rsp); // New substring address
3084 } // non constant
3085
3086 bind(CHECK_STR);
3087 cmpl(cnt1, stride);
3088 jccb(Assembler::aboveEqual, BIG_STRINGS);
3089
3090 // Check cross page boundary.
3091 movl(result, str1); // We need only low 32 bits
3092 andl(result, ((int)os::vm_page_size()-1));
3093 cmpl(result, ((int)os::vm_page_size()-16));
3094 jccb(Assembler::belowEqual, BIG_STRINGS);
3095
3096 subptr(rsp, 16);
3097 int stk_offset = -(1<<scale1);
3098 if (int_cnt2 < 0) { // not constant
3099 push(cnt2);
3100 stk_offset += wordSize;
3101 }
3102 movl(cnt2, cnt1);
3103
3104 bind(COPY_STR);
3105 if (ae == StrIntrinsicNode::LL) {
3106 load_unsigned_byte(result, Address(str1, cnt2, scale1, -1));
3107 movb(Address(rsp, cnt2, scale1, stk_offset), result);
3108 } else {
3109 load_unsigned_short(result, Address(str1, cnt2, scale1, -2));
3110 movw(Address(rsp, cnt2, scale1, stk_offset), result);
3111 }
3112 decrement(cnt2);
3113 jccb(Assembler::notZero, COPY_STR);
3114
3115 if (int_cnt2 < 0) { // not constant
3116 pop(cnt2);
3117 }
3118 movptr(str1, rsp); // New string address
3119
3120 bind(BIG_STRINGS);
3121 // Load substring.
3122 if (int_cnt2 < 0) { // -1
3123 if (ae == StrIntrinsicNode::UL) {
3124 pmovzxbw(vec, Address(str2, 0));
3125 } else {
3126 movdqu(vec, Address(str2, 0));
3127 }
3128 push(cnt2); // substr count
3129 push(str2); // substr addr
3130 push(str1); // string addr
3131 } else {
3132 // Small (< 8 chars) constant substrings are loaded already.
3133 movl(cnt2, int_cnt2);
3134 }
3135 push(tmp); // original SP
3136
3137 } // Finished loading
3138
3139 //========================================================
3140 // Start search
3141 //
3142
3143 movptr(result, str1); // string addr
3144
3145 if (int_cnt2 < 0) { // Only for non constant substring
3146 jmpb(SCAN_TO_SUBSTR);
3147
3148 // SP saved at sp+0
3149 // String saved at sp+1*wordSize
3150 // Substr saved at sp+2*wordSize
3151 // Substr count saved at sp+3*wordSize
3152
3153 // Reload substr for rescan, this code
3154 // is executed only for large substrings (> 8 chars)
3155 bind(RELOAD_SUBSTR);
3156 movptr(str2, Address(rsp, 2*wordSize));
3157 movl(cnt2, Address(rsp, 3*wordSize));
3158 if (ae == StrIntrinsicNode::UL) {
3159 pmovzxbw(vec, Address(str2, 0));
3160 } else {
3161 movdqu(vec, Address(str2, 0));
3162 }
3163 // We came here after the beginning of the substring was
3164 // matched but the rest of it was not so we need to search
3165 // again. Start from the next element after the previous match.
3166 subptr(str1, result); // Restore counter
3167 if (ae == StrIntrinsicNode::UU || ae == StrIntrinsicNode::UL) {
3168 shrl(str1, 1);
3169 }
3170 addl(cnt1, str1);
3171 decrementl(cnt1); // Shift to next element
3172 cmpl(cnt1, cnt2);
3173 jcc(Assembler::negative, RET_NOT_FOUND); // Left less then substring
3174
3175 addptr(result, (1<<scale1));
3176 } // non constant
3177
3178 // Scan string for start of substr in 16-byte vectors
3179 bind(SCAN_TO_SUBSTR);
3180 assert(cnt1 == rdx && cnt2 == rax && tmp == rcx, "pcmpestri");
3181 pcmpestri(vec, Address(result, 0), mode);
3182 jccb(Assembler::below, FOUND_CANDIDATE); // CF == 1
3183 subl(cnt1, stride);
3184 jccb(Assembler::lessEqual, RET_NOT_FOUND); // Scanned full string
3185 cmpl(cnt1, cnt2);
3186 jccb(Assembler::negative, RET_NOT_FOUND); // Left less then substring
3187 addptr(result, 16);
3188
3189 bind(ADJUST_STR);
3190 cmpl(cnt1, stride); // Do not read beyond string
3191 jccb(Assembler::greaterEqual, SCAN_TO_SUBSTR);
3192 // Back-up string to avoid reading beyond string.
3193 lea(result, Address(result, cnt1, scale1, -16));
3194 movl(cnt1, stride);
3195 jmpb(SCAN_TO_SUBSTR);
3196
3197 // Found a potential substr
3198 bind(FOUND_CANDIDATE);
3199 // After pcmpestri tmp(rcx) contains matched element index
3200
3201 // Make sure string is still long enough
3202 subl(cnt1, tmp);
3203 cmpl(cnt1, cnt2);
3204 jccb(Assembler::greaterEqual, FOUND_SUBSTR);
3205 // Left less then substring.
3206
3207 bind(RET_NOT_FOUND);
3208 movl(result, -1);
3209 jmp(CLEANUP);
3210
3211 bind(FOUND_SUBSTR);
3212 // Compute start addr of substr
3213 lea(result, Address(result, tmp, scale1));
3214 if (int_cnt2 > 0) { // Constant substring
3215 // Repeat search for small substring (< 8 chars)
3216 // from new point without reloading substring.
3217 // Have to check that we don't read beyond string.
3218 cmpl(tmp, stride-int_cnt2);
3219 jccb(Assembler::greater, ADJUST_STR);
3220 // Fall through if matched whole substring.
3221 } else { // non constant
3222 assert(int_cnt2 == -1, "should be != 0");
3223
3224 addl(tmp, cnt2);
3225 // Found result if we matched whole substring.
3226 cmpl(tmp, stride);
3227 jcc(Assembler::lessEqual, RET_FOUND);
3228
3229 // Repeat search for small substring (<= 8 chars)
3230 // from new point 'str1' without reloading substring.
3231 cmpl(cnt2, stride);
3232 // Have to check that we don't read beyond string.
3233 jccb(Assembler::lessEqual, ADJUST_STR);
3234
3235 Label CHECK_NEXT, CONT_SCAN_SUBSTR, RET_FOUND_LONG;
3236 // Compare the rest of substring (> 8 chars).
3237 movptr(str1, result);
3238
3239 cmpl(tmp, cnt2);
3240 // First 8 chars are already matched.
3241 jccb(Assembler::equal, CHECK_NEXT);
3242
3243 bind(SCAN_SUBSTR);
3244 pcmpestri(vec, Address(str1, 0), mode);
3245 // Need to reload strings pointers if not matched whole vector
3246 jcc(Assembler::noOverflow, RELOAD_SUBSTR); // OF == 0
3247
3248 bind(CHECK_NEXT);
3249 subl(cnt2, stride);
3250 jccb(Assembler::lessEqual, RET_FOUND_LONG); // Found full substring
3251 addptr(str1, 16);
3252 if (ae == StrIntrinsicNode::UL) {
3253 addptr(str2, 8);
3254 } else {
3255 addptr(str2, 16);
3256 }
3257 subl(cnt1, stride);
3258 cmpl(cnt2, stride); // Do not read beyond substring
3259 jccb(Assembler::greaterEqual, CONT_SCAN_SUBSTR);
3260 // Back-up strings to avoid reading beyond substring.
3261
3262 if (ae == StrIntrinsicNode::UL) {
3263 lea(str2, Address(str2, cnt2, scale2, -8));
3264 lea(str1, Address(str1, cnt2, scale1, -16));
3265 } else {
3266 lea(str2, Address(str2, cnt2, scale2, -16));
3267 lea(str1, Address(str1, cnt2, scale1, -16));
3268 }
3269 subl(cnt1, cnt2);
3270 movl(cnt2, stride);
3271 addl(cnt1, stride);
3272 bind(CONT_SCAN_SUBSTR);
3273 if (ae == StrIntrinsicNode::UL) {
3274 pmovzxbw(vec, Address(str2, 0));
3275 } else {
3276 movdqu(vec, Address(str2, 0));
3277 }
3278 jmp(SCAN_SUBSTR);
3279
3280 bind(RET_FOUND_LONG);
3281 movptr(str1, Address(rsp, wordSize));
3282 } // non constant
3283
3284 bind(RET_FOUND);
3285 // Compute substr offset
3286 subptr(result, str1);
3287 if (ae == StrIntrinsicNode::UU || ae == StrIntrinsicNode::UL) {
3288 shrl(result, 1); // index
3289 }
3290 bind(CLEANUP);
3291 pop(rsp); // restore SP
3292
3293 } // string_indexof
3294
3295 void C2_MacroAssembler::string_indexof_char(Register str1, Register cnt1, Register ch, Register result,
3296 XMMRegister vec1, XMMRegister vec2, XMMRegister vec3, Register tmp) {
3297 ShortBranchVerifier sbv(this);
3298 assert(UseSSE42Intrinsics, "SSE4.2 intrinsics are required");
3299
3300 int stride = 8;
3301
3302 Label FOUND_CHAR, SCAN_TO_CHAR, SCAN_TO_CHAR_LOOP,
3303 SCAN_TO_8_CHAR, SCAN_TO_8_CHAR_LOOP, SCAN_TO_16_CHAR_LOOP,
3304 RET_NOT_FOUND, SCAN_TO_8_CHAR_INIT,
3305 FOUND_SEQ_CHAR, DONE_LABEL;
3306
3307 movptr(result, str1);
3308 if (UseAVX >= 2) {
3309 cmpl(cnt1, stride);
3310 jcc(Assembler::less, SCAN_TO_CHAR);
3311 cmpl(cnt1, 2*stride);
3312 jcc(Assembler::less, SCAN_TO_8_CHAR_INIT);
3313 movdl(vec1, ch);
3314 vpbroadcastw(vec1, vec1, Assembler::AVX_256bit);
3315 vpxor(vec2, vec2);
3316 movl(tmp, cnt1);
3317 andl(tmp, 0xFFFFFFF0); //vector count (in chars)
3318 andl(cnt1,0x0000000F); //tail count (in chars)
3319
3320 bind(SCAN_TO_16_CHAR_LOOP);
3321 vmovdqu(vec3, Address(result, 0));
3322 vpcmpeqw(vec3, vec3, vec1, 1);
3323 vptest(vec2, vec3);
3324 jcc(Assembler::carryClear, FOUND_CHAR);
3325 addptr(result, 32);
3326 subl(tmp, 2*stride);
3327 jcc(Assembler::notZero, SCAN_TO_16_CHAR_LOOP);
3328 jmp(SCAN_TO_8_CHAR);
3329 bind(SCAN_TO_8_CHAR_INIT);
3330 movdl(vec1, ch);
3331 pshuflw(vec1, vec1, 0x00);
3332 pshufd(vec1, vec1, 0);
3333 pxor(vec2, vec2);
3334 }
3335 bind(SCAN_TO_8_CHAR);
3336 cmpl(cnt1, stride);
3337 jcc(Assembler::less, SCAN_TO_CHAR);
3338 if (UseAVX < 2) {
3339 movdl(vec1, ch);
3340 pshuflw(vec1, vec1, 0x00);
3341 pshufd(vec1, vec1, 0);
3342 pxor(vec2, vec2);
3343 }
3344 movl(tmp, cnt1);
3345 andl(tmp, 0xFFFFFFF8); //vector count (in chars)
3346 andl(cnt1,0x00000007); //tail count (in chars)
3347
3348 bind(SCAN_TO_8_CHAR_LOOP);
3349 movdqu(vec3, Address(result, 0));
3350 pcmpeqw(vec3, vec1);
3351 ptest(vec2, vec3);
3352 jcc(Assembler::carryClear, FOUND_CHAR);
3353 addptr(result, 16);
3354 subl(tmp, stride);
3355 jcc(Assembler::notZero, SCAN_TO_8_CHAR_LOOP);
3356 bind(SCAN_TO_CHAR);
3357 testl(cnt1, cnt1);
3358 jcc(Assembler::zero, RET_NOT_FOUND);
3359 bind(SCAN_TO_CHAR_LOOP);
3360 load_unsigned_short(tmp, Address(result, 0));
3361 cmpl(ch, tmp);
3362 jccb(Assembler::equal, FOUND_SEQ_CHAR);
3363 addptr(result, 2);
3364 subl(cnt1, 1);
3365 jccb(Assembler::zero, RET_NOT_FOUND);
3366 jmp(SCAN_TO_CHAR_LOOP);
3367
3368 bind(RET_NOT_FOUND);
3369 movl(result, -1);
3370 jmpb(DONE_LABEL);
3371
3372 bind(FOUND_CHAR);
3373 if (UseAVX >= 2) {
3374 vpmovmskb(tmp, vec3);
3375 } else {
3376 pmovmskb(tmp, vec3);
3377 }
3378 bsfl(ch, tmp);
3379 addptr(result, ch);
3380
3381 bind(FOUND_SEQ_CHAR);
3382 subptr(result, str1);
3383 shrl(result, 1);
3384
3385 bind(DONE_LABEL);
3386 } // string_indexof_char
3387
3388 void C2_MacroAssembler::stringL_indexof_char(Register str1, Register cnt1, Register ch, Register result,
3389 XMMRegister vec1, XMMRegister vec2, XMMRegister vec3, Register tmp) {
3390 ShortBranchVerifier sbv(this);
3391 assert(UseSSE42Intrinsics, "SSE4.2 intrinsics are required");
3392
3393 int stride = 16;
3394
3395 Label FOUND_CHAR, SCAN_TO_CHAR_INIT, SCAN_TO_CHAR_LOOP,
3396 SCAN_TO_16_CHAR, SCAN_TO_16_CHAR_LOOP, SCAN_TO_32_CHAR_LOOP,
3397 RET_NOT_FOUND, SCAN_TO_16_CHAR_INIT,
3398 FOUND_SEQ_CHAR, DONE_LABEL;
3399
3400 movptr(result, str1);
3401 if (UseAVX >= 2) {
3402 cmpl(cnt1, stride);
3403 jcc(Assembler::less, SCAN_TO_CHAR_INIT);
3404 cmpl(cnt1, stride*2);
3405 jcc(Assembler::less, SCAN_TO_16_CHAR_INIT);
3406 movdl(vec1, ch);
3407 vpbroadcastb(vec1, vec1, Assembler::AVX_256bit);
3408 vpxor(vec2, vec2);
3409 movl(tmp, cnt1);
3410 andl(tmp, 0xFFFFFFE0); //vector count (in chars)
3411 andl(cnt1,0x0000001F); //tail count (in chars)
3412
3413 bind(SCAN_TO_32_CHAR_LOOP);
3414 vmovdqu(vec3, Address(result, 0));
3415 vpcmpeqb(vec3, vec3, vec1, Assembler::AVX_256bit);
3416 vptest(vec2, vec3);
3417 jcc(Assembler::carryClear, FOUND_CHAR);
3418 addptr(result, 32);
3419 subl(tmp, stride*2);
3420 jcc(Assembler::notZero, SCAN_TO_32_CHAR_LOOP);
3421 jmp(SCAN_TO_16_CHAR);
3422
3423 bind(SCAN_TO_16_CHAR_INIT);
3424 movdl(vec1, ch);
3425 pxor(vec2, vec2);
3426 pshufb(vec1, vec2);
3427 }
3428
3429 bind(SCAN_TO_16_CHAR);
3430 cmpl(cnt1, stride);
3431 jcc(Assembler::less, SCAN_TO_CHAR_INIT);//less than 16 entries left
3432 if (UseAVX < 2) {
3433 movdl(vec1, ch);
3434 pxor(vec2, vec2);
3435 pshufb(vec1, vec2);
3436 }
3437 movl(tmp, cnt1);
3438 andl(tmp, 0xFFFFFFF0); //vector count (in bytes)
3439 andl(cnt1,0x0000000F); //tail count (in bytes)
3440
3441 bind(SCAN_TO_16_CHAR_LOOP);
3442 movdqu(vec3, Address(result, 0));
3443 pcmpeqb(vec3, vec1);
3444 ptest(vec2, vec3);
3445 jcc(Assembler::carryClear, FOUND_CHAR);
3446 addptr(result, 16);
3447 subl(tmp, stride);
3448 jcc(Assembler::notZero, SCAN_TO_16_CHAR_LOOP);//last 16 items...
3449
3450 bind(SCAN_TO_CHAR_INIT);
3451 testl(cnt1, cnt1);
3452 jcc(Assembler::zero, RET_NOT_FOUND);
3453 bind(SCAN_TO_CHAR_LOOP);
3454 load_unsigned_byte(tmp, Address(result, 0));
3455 cmpl(ch, tmp);
3456 jccb(Assembler::equal, FOUND_SEQ_CHAR);
3457 addptr(result, 1);
3458 subl(cnt1, 1);
3459 jccb(Assembler::zero, RET_NOT_FOUND);
3460 jmp(SCAN_TO_CHAR_LOOP);
3461
3462 bind(RET_NOT_FOUND);
3463 movl(result, -1);
3464 jmpb(DONE_LABEL);
3465
3466 bind(FOUND_CHAR);
3467 if (UseAVX >= 2) {
3468 vpmovmskb(tmp, vec3);
3469 } else {
3470 pmovmskb(tmp, vec3);
3471 }
3472 bsfl(ch, tmp);
3473 addptr(result, ch);
3474
3475 bind(FOUND_SEQ_CHAR);
3476 subptr(result, str1);
3477
3478 bind(DONE_LABEL);
3479 } // stringL_indexof_char
3480
3481 int C2_MacroAssembler::arrays_hashcode_elsize(BasicType eltype) {
3482 switch (eltype) {
3483 case T_BOOLEAN: return sizeof(jboolean);
3484 case T_BYTE: return sizeof(jbyte);
3485 case T_SHORT: return sizeof(jshort);
3486 case T_CHAR: return sizeof(jchar);
3487 case T_INT: return sizeof(jint);
3488 default:
3489 ShouldNotReachHere();
3490 return -1;
3491 }
3492 }
3493
3494 void C2_MacroAssembler::arrays_hashcode_elload(Register dst, Address src, BasicType eltype) {
3495 switch (eltype) {
3496 // T_BOOLEAN used as surrogate for unsigned byte
3497 case T_BOOLEAN: movzbl(dst, src); break;
3498 case T_BYTE: movsbl(dst, src); break;
3499 case T_SHORT: movswl(dst, src); break;
3500 case T_CHAR: movzwl(dst, src); break;
3501 case T_INT: movl(dst, src); break;
3502 default:
3503 ShouldNotReachHere();
3504 }
3505 }
3506
3507 void C2_MacroAssembler::arrays_hashcode_elvload(XMMRegister dst, Address src, BasicType eltype) {
3508 load_vector(eltype, dst, src, arrays_hashcode_elsize(eltype) * 8);
3509 }
3510
3511 void C2_MacroAssembler::arrays_hashcode_elvload(XMMRegister dst, AddressLiteral src, BasicType eltype) {
3512 load_vector(eltype, dst, src, arrays_hashcode_elsize(eltype) * 8);
3513 }
3514
3515 void C2_MacroAssembler::arrays_hashcode_elvcast(XMMRegister dst, BasicType eltype) {
3516 const int vlen = Assembler::AVX_256bit;
3517 switch (eltype) {
3518 case T_BOOLEAN: vector_unsigned_cast(dst, dst, vlen, T_BYTE, T_INT); break;
3519 case T_BYTE: vector_signed_cast(dst, dst, vlen, T_BYTE, T_INT); break;
3520 case T_SHORT: vector_signed_cast(dst, dst, vlen, T_SHORT, T_INT); break;
3521 case T_CHAR: vector_unsigned_cast(dst, dst, vlen, T_SHORT, T_INT); break;
3522 case T_INT:
3523 // do nothing
3524 break;
3525 default:
3526 ShouldNotReachHere();
3527 }
3528 }
3529
3530 void C2_MacroAssembler::arrays_hashcode(Register ary1, Register cnt1, Register result,
3531 Register index, Register tmp2, Register tmp3, XMMRegister vnext,
3532 XMMRegister vcoef0, XMMRegister vcoef1, XMMRegister vcoef2, XMMRegister vcoef3,
3533 XMMRegister vresult0, XMMRegister vresult1, XMMRegister vresult2, XMMRegister vresult3,
3534 XMMRegister vtmp0, XMMRegister vtmp1, XMMRegister vtmp2, XMMRegister vtmp3,
3535 BasicType eltype) {
3536 ShortBranchVerifier sbv(this);
3537 assert(UseAVX >= 2, "AVX2 intrinsics are required");
3538 assert_different_registers(ary1, cnt1, result, index, tmp2, tmp3);
3539 assert_different_registers(vnext, vcoef0, vcoef1, vcoef2, vcoef3, vresult0, vresult1, vresult2, vresult3, vtmp0, vtmp1, vtmp2, vtmp3);
3540
3541 Label SHORT_UNROLLED_BEGIN, SHORT_UNROLLED_LOOP_BEGIN,
3542 SHORT_UNROLLED_LOOP_EXIT,
3543 UNROLLED_SCALAR_LOOP_BEGIN, UNROLLED_SCALAR_SKIP, UNROLLED_SCALAR_RESUME,
3544 UNROLLED_VECTOR_LOOP_BEGIN,
3545 END;
3546 switch (eltype) {
3547 case T_BOOLEAN: BLOCK_COMMENT("arrays_hashcode(unsigned byte) {"); break;
3548 case T_CHAR: BLOCK_COMMENT("arrays_hashcode(char) {"); break;
3549 case T_BYTE: BLOCK_COMMENT("arrays_hashcode(byte) {"); break;
3550 case T_SHORT: BLOCK_COMMENT("arrays_hashcode(short) {"); break;
3551 case T_INT: BLOCK_COMMENT("arrays_hashcode(int) {"); break;
3552 default: BLOCK_COMMENT("arrays_hashcode {"); break;
3553 }
3554
3555 // For "renaming" for readibility of the code
3556 const XMMRegister vcoef[] = { vcoef0, vcoef1, vcoef2, vcoef3 },
3557 vresult[] = { vresult0, vresult1, vresult2, vresult3 },
3558 vtmp[] = { vtmp0, vtmp1, vtmp2, vtmp3 };
3559
3560 const int elsize = arrays_hashcode_elsize(eltype);
3561
3562 /*
3563 if (cnt1 >= 2) {
3564 if (cnt1 >= 32) {
3565 UNROLLED VECTOR LOOP
3566 }
3567 UNROLLED SCALAR LOOP
3568 }
3569 SINGLE SCALAR
3570 */
3571
3572 cmpl(cnt1, 32);
3573 jcc(Assembler::less, SHORT_UNROLLED_BEGIN);
3574
3575 // cnt1 >= 32 && generate_vectorized_loop
3576 xorl(index, index);
3577
3578 // vresult = IntVector.zero(I256);
3579 for (int idx = 0; idx < 4; idx++) {
3580 vpxor(vresult[idx], vresult[idx]);
3581 }
3582 // vnext = IntVector.broadcast(I256, power_of_31_backwards[0]);
3583 Register bound = tmp2;
3584 Register next = tmp3;
3585 lea(tmp2, ExternalAddress(StubRoutines::x86::arrays_hashcode_powers_of_31() + (0 * sizeof(jint))));
3586 movl(next, Address(tmp2, 0));
3587 movdl(vnext, next);
3588 vpbroadcastd(vnext, vnext, Assembler::AVX_256bit);
3589
3590 // index = 0;
3591 // bound = cnt1 & ~(32 - 1);
3592 movl(bound, cnt1);
3593 andl(bound, ~(32 - 1));
3594 // for (; index < bound; index += 32) {
3595 bind(UNROLLED_VECTOR_LOOP_BEGIN);
3596 // result *= next;
3597 imull(result, next);
3598 // loop fission to upfront the cost of fetching from memory, OOO execution
3599 // can then hopefully do a better job of prefetching
3600 for (int idx = 0; idx < 4; idx++) {
3601 arrays_hashcode_elvload(vtmp[idx], Address(ary1, index, Address::times(elsize), 8 * idx * elsize), eltype);
3602 }
3603 // vresult = vresult * vnext + ary1[index+8*idx:index+8*idx+7];
3604 for (int idx = 0; idx < 4; idx++) {
3605 vpmulld(vresult[idx], vresult[idx], vnext, Assembler::AVX_256bit);
3606 arrays_hashcode_elvcast(vtmp[idx], eltype);
3607 vpaddd(vresult[idx], vresult[idx], vtmp[idx], Assembler::AVX_256bit);
3608 }
3609 // index += 32;
3610 addl(index, 32);
3611 // index < bound;
3612 cmpl(index, bound);
3613 jcc(Assembler::less, UNROLLED_VECTOR_LOOP_BEGIN);
3614 // }
3615
3616 lea(ary1, Address(ary1, bound, Address::times(elsize)));
3617 subl(cnt1, bound);
3618 // release bound
3619
3620 // vresult *= IntVector.fromArray(I256, power_of_31_backwards, 1);
3621 for (int idx = 0; idx < 4; idx++) {
3622 lea(tmp2, ExternalAddress(StubRoutines::x86::arrays_hashcode_powers_of_31() + ((8 * idx + 1) * sizeof(jint))));
3623 arrays_hashcode_elvload(vcoef[idx], Address(tmp2, 0), T_INT);
3624 vpmulld(vresult[idx], vresult[idx], vcoef[idx], Assembler::AVX_256bit);
3625 }
3626 // result += vresult.reduceLanes(ADD);
3627 for (int idx = 0; idx < 4; idx++) {
3628 reduceI(Op_AddReductionVI, 256/(sizeof(jint) * 8), result, result, vresult[idx], vtmp[(idx * 2 + 0) % 4], vtmp[(idx * 2 + 1) % 4]);
3629 }
3630
3631 // } else if (cnt1 < 32) {
3632
3633 bind(SHORT_UNROLLED_BEGIN);
3634 // int i = 1;
3635 movl(index, 1);
3636 cmpl(index, cnt1);
3637 jcc(Assembler::greaterEqual, SHORT_UNROLLED_LOOP_EXIT);
3638
3639 // for (; i < cnt1 ; i += 2) {
3640 bind(SHORT_UNROLLED_LOOP_BEGIN);
3641 movl(tmp3, 961);
3642 imull(result, tmp3);
3643 arrays_hashcode_elload(tmp2, Address(ary1, index, Address::times(elsize), -elsize), eltype);
3644 movl(tmp3, tmp2);
3645 shll(tmp3, 5);
3646 subl(tmp3, tmp2);
3647 addl(result, tmp3);
3648 arrays_hashcode_elload(tmp3, Address(ary1, index, Address::times(elsize)), eltype);
3649 addl(result, tmp3);
3650 addl(index, 2);
3651 cmpl(index, cnt1);
3652 jccb(Assembler::less, SHORT_UNROLLED_LOOP_BEGIN);
3653
3654 // }
3655 // if (i >= cnt1) {
3656 bind(SHORT_UNROLLED_LOOP_EXIT);
3657 jccb(Assembler::greater, END);
3658 movl(tmp2, result);
3659 shll(result, 5);
3660 subl(result, tmp2);
3661 arrays_hashcode_elload(tmp3, Address(ary1, index, Address::times(elsize), -elsize), eltype);
3662 addl(result, tmp3);
3663 // }
3664 bind(END);
3665
3666 BLOCK_COMMENT("} // arrays_hashcode");
3667
3668 } // arrays_hashcode
3669
3670 // helper function for string_compare
3671 void C2_MacroAssembler::load_next_elements(Register elem1, Register elem2, Register str1, Register str2,
3672 Address::ScaleFactor scale, Address::ScaleFactor scale1,
3673 Address::ScaleFactor scale2, Register index, int ae) {
3674 if (ae == StrIntrinsicNode::LL) {
3675 load_unsigned_byte(elem1, Address(str1, index, scale, 0));
3676 load_unsigned_byte(elem2, Address(str2, index, scale, 0));
3677 } else if (ae == StrIntrinsicNode::UU) {
3678 load_unsigned_short(elem1, Address(str1, index, scale, 0));
3679 load_unsigned_short(elem2, Address(str2, index, scale, 0));
3680 } else {
3681 load_unsigned_byte(elem1, Address(str1, index, scale1, 0));
3682 load_unsigned_short(elem2, Address(str2, index, scale2, 0));
3683 }
3684 }
3685
3686 // Compare strings, used for char[] and byte[].
3687 void C2_MacroAssembler::string_compare(Register str1, Register str2,
3688 Register cnt1, Register cnt2, Register result,
3689 XMMRegister vec1, int ae, KRegister mask) {
3690 ShortBranchVerifier sbv(this);
3691 Label LENGTH_DIFF_LABEL, POP_LABEL, DONE_LABEL, WHILE_HEAD_LABEL;
3692 Label COMPARE_WIDE_VECTORS_LOOP_FAILED; // used only AVX3
3693 int stride, stride2, adr_stride, adr_stride1, adr_stride2;
3694 int stride2x2 = 0x40;
3695 Address::ScaleFactor scale = Address::no_scale;
3696 Address::ScaleFactor scale1 = Address::no_scale;
3697 Address::ScaleFactor scale2 = Address::no_scale;
3698
3699 if (ae != StrIntrinsicNode::LL) {
3700 stride2x2 = 0x20;
3701 }
3702
3703 if (ae == StrIntrinsicNode::LU || ae == StrIntrinsicNode::UL) {
3704 shrl(cnt2, 1);
3705 }
3706 // Compute the minimum of the string lengths and the
3707 // difference of the string lengths (stack).
3708 // Do the conditional move stuff
3709 movl(result, cnt1);
3710 subl(cnt1, cnt2);
3711 push(cnt1);
3712 cmov32(Assembler::lessEqual, cnt2, result); // cnt2 = min(cnt1, cnt2)
3713
3714 // Is the minimum length zero?
3715 testl(cnt2, cnt2);
3716 jcc(Assembler::zero, LENGTH_DIFF_LABEL);
3717 if (ae == StrIntrinsicNode::LL) {
3718 // Load first bytes
3719 load_unsigned_byte(result, Address(str1, 0)); // result = str1[0]
3720 load_unsigned_byte(cnt1, Address(str2, 0)); // cnt1 = str2[0]
3721 } else if (ae == StrIntrinsicNode::UU) {
3722 // Load first characters
3723 load_unsigned_short(result, Address(str1, 0));
3724 load_unsigned_short(cnt1, Address(str2, 0));
3725 } else {
3726 load_unsigned_byte(result, Address(str1, 0));
3727 load_unsigned_short(cnt1, Address(str2, 0));
3728 }
3729 subl(result, cnt1);
3730 jcc(Assembler::notZero, POP_LABEL);
3731
3732 if (ae == StrIntrinsicNode::UU) {
3733 // Divide length by 2 to get number of chars
3734 shrl(cnt2, 1);
3735 }
3736 cmpl(cnt2, 1);
3737 jcc(Assembler::equal, LENGTH_DIFF_LABEL);
3738
3739 // Check if the strings start at the same location and setup scale and stride
3740 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3741 cmpptr(str1, str2);
3742 jcc(Assembler::equal, LENGTH_DIFF_LABEL);
3743 if (ae == StrIntrinsicNode::LL) {
3744 scale = Address::times_1;
3745 stride = 16;
3746 } else {
3747 scale = Address::times_2;
3748 stride = 8;
3749 }
3750 } else {
3751 scale1 = Address::times_1;
3752 scale2 = Address::times_2;
3753 // scale not used
3754 stride = 8;
3755 }
3756
3757 if (UseAVX >= 2 && UseSSE42Intrinsics) {
3758 Label COMPARE_WIDE_VECTORS, VECTOR_NOT_EQUAL, COMPARE_WIDE_TAIL, COMPARE_SMALL_STR;
3759 Label COMPARE_WIDE_VECTORS_LOOP, COMPARE_16_CHARS, COMPARE_INDEX_CHAR;
3760 Label COMPARE_WIDE_VECTORS_LOOP_AVX2;
3761 Label COMPARE_TAIL_LONG;
3762 Label COMPARE_WIDE_VECTORS_LOOP_AVX3; // used only AVX3
3763
3764 int pcmpmask = 0x19;
3765 if (ae == StrIntrinsicNode::LL) {
3766 pcmpmask &= ~0x01;
3767 }
3768
3769 // Setup to compare 16-chars (32-bytes) vectors,
3770 // start from first character again because it has aligned address.
3771 if (ae == StrIntrinsicNode::LL) {
3772 stride2 = 32;
3773 } else {
3774 stride2 = 16;
3775 }
3776 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3777 adr_stride = stride << scale;
3778 } else {
3779 adr_stride1 = 8; //stride << scale1;
3780 adr_stride2 = 16; //stride << scale2;
3781 }
3782
3783 assert(result == rax && cnt2 == rdx && cnt1 == rcx, "pcmpestri");
3784 // rax and rdx are used by pcmpestri as elements counters
3785 movl(result, cnt2);
3786 andl(cnt2, ~(stride2-1)); // cnt2 holds the vector count
3787 jcc(Assembler::zero, COMPARE_TAIL_LONG);
3788
3789 // fast path : compare first 2 8-char vectors.
3790 bind(COMPARE_16_CHARS);
3791 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3792 movdqu(vec1, Address(str1, 0));
3793 } else {
3794 pmovzxbw(vec1, Address(str1, 0));
3795 }
3796 pcmpestri(vec1, Address(str2, 0), pcmpmask);
3797 jccb(Assembler::below, COMPARE_INDEX_CHAR);
3798
3799 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3800 movdqu(vec1, Address(str1, adr_stride));
3801 pcmpestri(vec1, Address(str2, adr_stride), pcmpmask);
3802 } else {
3803 pmovzxbw(vec1, Address(str1, adr_stride1));
3804 pcmpestri(vec1, Address(str2, adr_stride2), pcmpmask);
3805 }
3806 jccb(Assembler::aboveEqual, COMPARE_WIDE_VECTORS);
3807 addl(cnt1, stride);
3808
3809 // Compare the characters at index in cnt1
3810 bind(COMPARE_INDEX_CHAR); // cnt1 has the offset of the mismatching character
3811 load_next_elements(result, cnt2, str1, str2, scale, scale1, scale2, cnt1, ae);
3812 subl(result, cnt2);
3813 jmp(POP_LABEL);
3814
3815 // Setup the registers to start vector comparison loop
3816 bind(COMPARE_WIDE_VECTORS);
3817 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3818 lea(str1, Address(str1, result, scale));
3819 lea(str2, Address(str2, result, scale));
3820 } else {
3821 lea(str1, Address(str1, result, scale1));
3822 lea(str2, Address(str2, result, scale2));
3823 }
3824 subl(result, stride2);
3825 subl(cnt2, stride2);
3826 jcc(Assembler::zero, COMPARE_WIDE_TAIL);
3827 negptr(result);
3828
3829 // In a loop, compare 16-chars (32-bytes) at once using (vpxor+vptest)
3830 bind(COMPARE_WIDE_VECTORS_LOOP);
3831
3832 if ((AVX3Threshold == 0) && VM_Version::supports_avx512vlbw()) { // trying 64 bytes fast loop
3833 cmpl(cnt2, stride2x2);
3834 jccb(Assembler::below, COMPARE_WIDE_VECTORS_LOOP_AVX2);
3835 testl(cnt2, stride2x2-1); // cnt2 holds the vector count
3836 jccb(Assembler::notZero, COMPARE_WIDE_VECTORS_LOOP_AVX2); // means we cannot subtract by 0x40
3837
3838 bind(COMPARE_WIDE_VECTORS_LOOP_AVX3); // the hottest loop
3839 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3840 evmovdquq(vec1, Address(str1, result, scale), Assembler::AVX_512bit);
3841 evpcmpeqb(mask, vec1, Address(str2, result, scale), Assembler::AVX_512bit); // k7 == 11..11, if operands equal, otherwise k7 has some 0
3842 } else {
3843 vpmovzxbw(vec1, Address(str1, result, scale1), Assembler::AVX_512bit);
3844 evpcmpeqb(mask, vec1, Address(str2, result, scale2), Assembler::AVX_512bit); // k7 == 11..11, if operands equal, otherwise k7 has some 0
3845 }
3846 kortestql(mask, mask);
3847 jcc(Assembler::aboveEqual, COMPARE_WIDE_VECTORS_LOOP_FAILED); // miscompare
3848 addptr(result, stride2x2); // update since we already compared at this addr
3849 subl(cnt2, stride2x2); // and sub the size too
3850 jccb(Assembler::notZero, COMPARE_WIDE_VECTORS_LOOP_AVX3);
3851
3852 vpxor(vec1, vec1);
3853 jmpb(COMPARE_WIDE_TAIL);
3854 }//if (VM_Version::supports_avx512vlbw())
3855
3856 bind(COMPARE_WIDE_VECTORS_LOOP_AVX2);
3857 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3858 vmovdqu(vec1, Address(str1, result, scale));
3859 vpxor(vec1, Address(str2, result, scale));
3860 } else {
3861 vpmovzxbw(vec1, Address(str1, result, scale1), Assembler::AVX_256bit);
3862 vpxor(vec1, Address(str2, result, scale2));
3863 }
3864 vptest(vec1, vec1);
3865 jcc(Assembler::notZero, VECTOR_NOT_EQUAL);
3866 addptr(result, stride2);
3867 subl(cnt2, stride2);
3868 jcc(Assembler::notZero, COMPARE_WIDE_VECTORS_LOOP);
3869 // clean upper bits of YMM registers
3870 vpxor(vec1, vec1);
3871
3872 // compare wide vectors tail
3873 bind(COMPARE_WIDE_TAIL);
3874 testptr(result, result);
3875 jcc(Assembler::zero, LENGTH_DIFF_LABEL);
3876
3877 movl(result, stride2);
3878 movl(cnt2, result);
3879 negptr(result);
3880 jmp(COMPARE_WIDE_VECTORS_LOOP_AVX2);
3881
3882 // Identifies the mismatching (higher or lower)16-bytes in the 32-byte vectors.
3883 bind(VECTOR_NOT_EQUAL);
3884 // clean upper bits of YMM registers
3885 vpxor(vec1, vec1);
3886 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3887 lea(str1, Address(str1, result, scale));
3888 lea(str2, Address(str2, result, scale));
3889 } else {
3890 lea(str1, Address(str1, result, scale1));
3891 lea(str2, Address(str2, result, scale2));
3892 }
3893 jmp(COMPARE_16_CHARS);
3894
3895 // Compare tail chars, length between 1 to 15 chars
3896 bind(COMPARE_TAIL_LONG);
3897 movl(cnt2, result);
3898 cmpl(cnt2, stride);
3899 jcc(Assembler::less, COMPARE_SMALL_STR);
3900
3901 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3902 movdqu(vec1, Address(str1, 0));
3903 } else {
3904 pmovzxbw(vec1, Address(str1, 0));
3905 }
3906 pcmpestri(vec1, Address(str2, 0), pcmpmask);
3907 jcc(Assembler::below, COMPARE_INDEX_CHAR);
3908 subptr(cnt2, stride);
3909 jcc(Assembler::zero, LENGTH_DIFF_LABEL);
3910 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3911 lea(str1, Address(str1, result, scale));
3912 lea(str2, Address(str2, result, scale));
3913 } else {
3914 lea(str1, Address(str1, result, scale1));
3915 lea(str2, Address(str2, result, scale2));
3916 }
3917 negptr(cnt2);
3918 jmpb(WHILE_HEAD_LABEL);
3919
3920 bind(COMPARE_SMALL_STR);
3921 } else if (UseSSE42Intrinsics) {
3922 Label COMPARE_WIDE_VECTORS, VECTOR_NOT_EQUAL, COMPARE_TAIL;
3923 int pcmpmask = 0x19;
3924 // Setup to compare 8-char (16-byte) vectors,
3925 // start from first character again because it has aligned address.
3926 movl(result, cnt2);
3927 andl(cnt2, ~(stride - 1)); // cnt2 holds the vector count
3928 if (ae == StrIntrinsicNode::LL) {
3929 pcmpmask &= ~0x01;
3930 }
3931 jcc(Assembler::zero, COMPARE_TAIL);
3932 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3933 lea(str1, Address(str1, result, scale));
3934 lea(str2, Address(str2, result, scale));
3935 } else {
3936 lea(str1, Address(str1, result, scale1));
3937 lea(str2, Address(str2, result, scale2));
3938 }
3939 negptr(result);
3940
3941 // pcmpestri
3942 // inputs:
3943 // vec1- substring
3944 // rax - negative string length (elements count)
3945 // mem - scanned string
3946 // rdx - string length (elements count)
3947 // pcmpmask - cmp mode: 11000 (string compare with negated result)
3948 // + 00 (unsigned bytes) or + 01 (unsigned shorts)
3949 // outputs:
3950 // rcx - first mismatched element index
3951 assert(result == rax && cnt2 == rdx && cnt1 == rcx, "pcmpestri");
3952
3953 bind(COMPARE_WIDE_VECTORS);
3954 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3955 movdqu(vec1, Address(str1, result, scale));
3956 pcmpestri(vec1, Address(str2, result, scale), pcmpmask);
3957 } else {
3958 pmovzxbw(vec1, Address(str1, result, scale1));
3959 pcmpestri(vec1, Address(str2, result, scale2), pcmpmask);
3960 }
3961 // After pcmpestri cnt1(rcx) contains mismatched element index
3962
3963 jccb(Assembler::below, VECTOR_NOT_EQUAL); // CF==1
3964 addptr(result, stride);
3965 subptr(cnt2, stride);
3966 jccb(Assembler::notZero, COMPARE_WIDE_VECTORS);
3967
3968 // compare wide vectors tail
3969 testptr(result, result);
3970 jcc(Assembler::zero, LENGTH_DIFF_LABEL);
3971
3972 movl(cnt2, stride);
3973 movl(result, stride);
3974 negptr(result);
3975 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3976 movdqu(vec1, Address(str1, result, scale));
3977 pcmpestri(vec1, Address(str2, result, scale), pcmpmask);
3978 } else {
3979 pmovzxbw(vec1, Address(str1, result, scale1));
3980 pcmpestri(vec1, Address(str2, result, scale2), pcmpmask);
3981 }
3982 jccb(Assembler::aboveEqual, LENGTH_DIFF_LABEL);
3983
3984 // Mismatched characters in the vectors
3985 bind(VECTOR_NOT_EQUAL);
3986 addptr(cnt1, result);
3987 load_next_elements(result, cnt2, str1, str2, scale, scale1, scale2, cnt1, ae);
3988 subl(result, cnt2);
3989 jmpb(POP_LABEL);
3990
3991 bind(COMPARE_TAIL); // limit is zero
3992 movl(cnt2, result);
3993 // Fallthru to tail compare
3994 }
3995 // Shift str2 and str1 to the end of the arrays, negate min
3996 if (ae == StrIntrinsicNode::LL || ae == StrIntrinsicNode::UU) {
3997 lea(str1, Address(str1, cnt2, scale));
3998 lea(str2, Address(str2, cnt2, scale));
3999 } else {
4000 lea(str1, Address(str1, cnt2, scale1));
4001 lea(str2, Address(str2, cnt2, scale2));
4002 }
4003 decrementl(cnt2); // first character was compared already
4004 negptr(cnt2);
4005
4006 // Compare the rest of the elements
4007 bind(WHILE_HEAD_LABEL);
4008 load_next_elements(result, cnt1, str1, str2, scale, scale1, scale2, cnt2, ae);
4009 subl(result, cnt1);
4010 jccb(Assembler::notZero, POP_LABEL);
4011 increment(cnt2);
4012 jccb(Assembler::notZero, WHILE_HEAD_LABEL);
4013
4014 // Strings are equal up to min length. Return the length difference.
4015 bind(LENGTH_DIFF_LABEL);
4016 pop(result);
4017 if (ae == StrIntrinsicNode::UU) {
4018 // Divide diff by 2 to get number of chars
4019 sarl(result, 1);
4020 }
4021 jmpb(DONE_LABEL);
4022
4023 if (VM_Version::supports_avx512vlbw()) {
4024
4025 bind(COMPARE_WIDE_VECTORS_LOOP_FAILED);
4026
4027 kmovql(cnt1, mask);
4028 notq(cnt1);
4029 bsfq(cnt2, cnt1);
4030 if (ae != StrIntrinsicNode::LL) {
4031 // Divide diff by 2 to get number of chars
4032 sarl(cnt2, 1);
4033 }
4034 addq(result, cnt2);
4035 if (ae == StrIntrinsicNode::LL) {
4036 load_unsigned_byte(cnt1, Address(str2, result));
4037 load_unsigned_byte(result, Address(str1, result));
4038 } else if (ae == StrIntrinsicNode::UU) {
4039 load_unsigned_short(cnt1, Address(str2, result, scale));
4040 load_unsigned_short(result, Address(str1, result, scale));
4041 } else {
4042 load_unsigned_short(cnt1, Address(str2, result, scale2));
4043 load_unsigned_byte(result, Address(str1, result, scale1));
4044 }
4045 subl(result, cnt1);
4046 jmpb(POP_LABEL);
4047 }//if (VM_Version::supports_avx512vlbw())
4048
4049 // Discard the stored length difference
4050 bind(POP_LABEL);
4051 pop(cnt1);
4052
4053 // That's it
4054 bind(DONE_LABEL);
4055 if(ae == StrIntrinsicNode::UL) {
4056 negl(result);
4057 }
4058
4059 }
4060
4061 // Search for Non-ASCII character (Negative byte value) in a byte array,
4062 // return the index of the first such character, otherwise the length
4063 // of the array segment searched.
4064 // ..\jdk\src\java.base\share\classes\java\lang\StringCoding.java
4065 // @IntrinsicCandidate
4066 // public static int countPositives(byte[] ba, int off, int len) {
4067 // for (int i = off; i < off + len; i++) {
4068 // if (ba[i] < 0) {
4069 // return i - off;
4070 // }
4071 // }
4072 // return len;
4073 // }
4074 void C2_MacroAssembler::count_positives(Register ary1, Register len,
4075 Register result, Register tmp1,
4076 XMMRegister vec1, XMMRegister vec2, KRegister mask1, KRegister mask2) {
4077 // rsi: byte array
4078 // rcx: len
4079 // rax: result
4080 ShortBranchVerifier sbv(this);
4081 assert_different_registers(ary1, len, result, tmp1);
4082 assert_different_registers(vec1, vec2);
4083 Label ADJUST, TAIL_ADJUST, DONE, TAIL_START, CHAR_ADJUST, COMPARE_CHAR, COMPARE_VECTORS, COMPARE_BYTE;
4084
4085 movl(result, len); // copy
4086 // len == 0
4087 testl(len, len);
4088 jcc(Assembler::zero, DONE);
4089
4090 if ((AVX3Threshold == 0) && (UseAVX > 2) && // AVX512
4091 VM_Version::supports_avx512vlbw() &&
4092 VM_Version::supports_bmi2()) {
4093
4094 Label test_64_loop, test_tail, BREAK_LOOP;
4095 movl(tmp1, len);
4096 vpxor(vec2, vec2, vec2, Assembler::AVX_512bit);
4097
4098 andl(tmp1, 0x0000003f); // tail count (in chars) 0x3F
4099 andl(len, 0xffffffc0); // vector count (in chars)
4100 jccb(Assembler::zero, test_tail);
4101
4102 lea(ary1, Address(ary1, len, Address::times_1));
4103 negptr(len);
4104
4105 bind(test_64_loop);
4106 // Check whether our 64 elements of size byte contain negatives
4107 evpcmpgtb(mask1, vec2, Address(ary1, len, Address::times_1), Assembler::AVX_512bit);
4108 kortestql(mask1, mask1);
4109 jcc(Assembler::notZero, BREAK_LOOP);
4110
4111 addptr(len, 64);
4112 jccb(Assembler::notZero, test_64_loop);
4113
4114 bind(test_tail);
4115 // bail out when there is nothing to be done
4116 testl(tmp1, -1);
4117 jcc(Assembler::zero, DONE);
4118
4119
4120 // check the tail for absense of negatives
4121 // ~(~0 << len) applied up to two times (for 32-bit scenario)
4122 {
4123 Register tmp3_aliased = len;
4124 mov64(tmp3_aliased, 0xFFFFFFFFFFFFFFFF);
4125 shlxq(tmp3_aliased, tmp3_aliased, tmp1);
4126 notq(tmp3_aliased);
4127 kmovql(mask2, tmp3_aliased);
4128 }
4129
4130 evpcmpgtb(mask1, mask2, vec2, Address(ary1, 0), Assembler::AVX_512bit);
4131 ktestq(mask1, mask2);
4132 jcc(Assembler::zero, DONE);
4133
4134 // do a full check for negative registers in the tail
4135 movl(len, tmp1); // tmp1 holds low 6-bit from original len;
4136 // ary1 already pointing to the right place
4137 jmpb(TAIL_START);
4138
4139 bind(BREAK_LOOP);
4140 // At least one byte in the last 64 byte block was negative.
4141 // Set up to look at the last 64 bytes as if they were a tail
4142 lea(ary1, Address(ary1, len, Address::times_1));
4143 addptr(result, len);
4144 // Ignore the very last byte: if all others are positive,
4145 // it must be negative, so we can skip right to the 2+1 byte
4146 // end comparison at this point
4147 orl(result, 63);
4148 movl(len, 63);
4149 // Fallthru to tail compare
4150 } else {
4151
4152 if (UseAVX >= 2) {
4153 // With AVX2, use 32-byte vector compare
4154 Label COMPARE_WIDE_VECTORS, BREAK_LOOP;
4155
4156 // Compare 32-byte vectors
4157 testl(len, 0xffffffe0); // vector count (in bytes)
4158 jccb(Assembler::zero, TAIL_START);
4159
4160 andl(len, 0xffffffe0);
4161 lea(ary1, Address(ary1, len, Address::times_1));
4162 negptr(len);
4163
4164 movl(tmp1, 0x80808080); // create mask to test for Unicode chars in vector
4165 movdl(vec2, tmp1);
4166 vpbroadcastd(vec2, vec2, Assembler::AVX_256bit);
4167
4168 bind(COMPARE_WIDE_VECTORS);
4169 vmovdqu(vec1, Address(ary1, len, Address::times_1));
4170 vptest(vec1, vec2);
4171 jccb(Assembler::notZero, BREAK_LOOP);
4172 addptr(len, 32);
4173 jccb(Assembler::notZero, COMPARE_WIDE_VECTORS);
4174
4175 testl(result, 0x0000001f); // any bytes remaining?
4176 jcc(Assembler::zero, DONE);
4177
4178 // Quick test using the already prepared vector mask
4179 movl(len, result);
4180 andl(len, 0x0000001f);
4181 vmovdqu(vec1, Address(ary1, len, Address::times_1, -32));
4182 vptest(vec1, vec2);
4183 jcc(Assembler::zero, DONE);
4184 // There are zeros, jump to the tail to determine exactly where
4185 jmpb(TAIL_START);
4186
4187 bind(BREAK_LOOP);
4188 // At least one byte in the last 32-byte vector is negative.
4189 // Set up to look at the last 32 bytes as if they were a tail
4190 lea(ary1, Address(ary1, len, Address::times_1));
4191 addptr(result, len);
4192 // Ignore the very last byte: if all others are positive,
4193 // it must be negative, so we can skip right to the 2+1 byte
4194 // end comparison at this point
4195 orl(result, 31);
4196 movl(len, 31);
4197 // Fallthru to tail compare
4198 } else if (UseSSE42Intrinsics) {
4199 // With SSE4.2, use double quad vector compare
4200 Label COMPARE_WIDE_VECTORS, BREAK_LOOP;
4201
4202 // Compare 16-byte vectors
4203 testl(len, 0xfffffff0); // vector count (in bytes)
4204 jcc(Assembler::zero, TAIL_START);
4205
4206 andl(len, 0xfffffff0);
4207 lea(ary1, Address(ary1, len, Address::times_1));
4208 negptr(len);
4209
4210 movl(tmp1, 0x80808080);
4211 movdl(vec2, tmp1);
4212 pshufd(vec2, vec2, 0);
4213
4214 bind(COMPARE_WIDE_VECTORS);
4215 movdqu(vec1, Address(ary1, len, Address::times_1));
4216 ptest(vec1, vec2);
4217 jccb(Assembler::notZero, BREAK_LOOP);
4218 addptr(len, 16);
4219 jccb(Assembler::notZero, COMPARE_WIDE_VECTORS);
4220
4221 testl(result, 0x0000000f); // len is zero, any bytes remaining?
4222 jcc(Assembler::zero, DONE);
4223
4224 // Quick test using the already prepared vector mask
4225 movl(len, result);
4226 andl(len, 0x0000000f); // tail count (in bytes)
4227 movdqu(vec1, Address(ary1, len, Address::times_1, -16));
4228 ptest(vec1, vec2);
4229 jcc(Assembler::zero, DONE);
4230 jmpb(TAIL_START);
4231
4232 bind(BREAK_LOOP);
4233 // At least one byte in the last 16-byte vector is negative.
4234 // Set up and look at the last 16 bytes as if they were a tail
4235 lea(ary1, Address(ary1, len, Address::times_1));
4236 addptr(result, len);
4237 // Ignore the very last byte: if all others are positive,
4238 // it must be negative, so we can skip right to the 2+1 byte
4239 // end comparison at this point
4240 orl(result, 15);
4241 movl(len, 15);
4242 // Fallthru to tail compare
4243 }
4244 }
4245
4246 bind(TAIL_START);
4247 // Compare 4-byte vectors
4248 andl(len, 0xfffffffc); // vector count (in bytes)
4249 jccb(Assembler::zero, COMPARE_CHAR);
4250
4251 lea(ary1, Address(ary1, len, Address::times_1));
4252 negptr(len);
4253
4254 bind(COMPARE_VECTORS);
4255 movl(tmp1, Address(ary1, len, Address::times_1));
4256 andl(tmp1, 0x80808080);
4257 jccb(Assembler::notZero, TAIL_ADJUST);
4258 addptr(len, 4);
4259 jccb(Assembler::notZero, COMPARE_VECTORS);
4260
4261 // Compare trailing char (final 2-3 bytes), if any
4262 bind(COMPARE_CHAR);
4263
4264 testl(result, 0x2); // tail char
4265 jccb(Assembler::zero, COMPARE_BYTE);
4266 load_unsigned_short(tmp1, Address(ary1, 0));
4267 andl(tmp1, 0x00008080);
4268 jccb(Assembler::notZero, CHAR_ADJUST);
4269 lea(ary1, Address(ary1, 2));
4270
4271 bind(COMPARE_BYTE);
4272 testl(result, 0x1); // tail byte
4273 jccb(Assembler::zero, DONE);
4274 load_unsigned_byte(tmp1, Address(ary1, 0));
4275 testl(tmp1, 0x00000080);
4276 jccb(Assembler::zero, DONE);
4277 subptr(result, 1);
4278 jmpb(DONE);
4279
4280 bind(TAIL_ADJUST);
4281 // there are negative bits in the last 4 byte block.
4282 // Adjust result and check the next three bytes
4283 addptr(result, len);
4284 orl(result, 3);
4285 lea(ary1, Address(ary1, len, Address::times_1));
4286 jmpb(COMPARE_CHAR);
4287
4288 bind(CHAR_ADJUST);
4289 // We are looking at a char + optional byte tail, and found that one
4290 // of the bytes in the char is negative. Adjust the result, check the
4291 // first byte and readjust if needed.
4292 andl(result, 0xfffffffc);
4293 testl(tmp1, 0x00000080); // little-endian, so lowest byte comes first
4294 jccb(Assembler::notZero, DONE);
4295 addptr(result, 1);
4296
4297 // That's it
4298 bind(DONE);
4299 if (UseAVX >= 2) {
4300 // clean upper bits of YMM registers
4301 vpxor(vec1, vec1);
4302 vpxor(vec2, vec2);
4303 }
4304 }
4305
4306 // Compare char[] or byte[] arrays aligned to 4 bytes or substrings.
4307 void C2_MacroAssembler::arrays_equals(bool is_array_equ, Register ary1, Register ary2,
4308 Register limit, Register result, Register chr,
4309 XMMRegister vec1, XMMRegister vec2, bool is_char,
4310 KRegister mask, bool expand_ary2) {
4311 // for expand_ary2, limit is the (smaller) size of the second array.
4312 ShortBranchVerifier sbv(this);
4313 Label TRUE_LABEL, FALSE_LABEL, DONE, COMPARE_VECTORS, COMPARE_CHAR, COMPARE_BYTE;
4314
4315 assert((!expand_ary2) || ((expand_ary2) && (UseAVX == 2)),
4316 "Expansion only implemented for AVX2");
4317
4318 int length_offset = arrayOopDesc::length_offset_in_bytes();
4319 int base_offset = arrayOopDesc::base_offset_in_bytes(is_char ? T_CHAR : T_BYTE);
4320
4321 Address::ScaleFactor scaleFactor = expand_ary2 ? Address::times_2 : Address::times_1;
4322 int scaleIncr = expand_ary2 ? 8 : 16;
4323
4324 if (is_array_equ) {
4325 // Check the input args
4326 cmpoop(ary1, ary2);
4327 jcc(Assembler::equal, TRUE_LABEL);
4328
4329 // Need additional checks for arrays_equals.
4330 testptr(ary1, ary1);
4331 jcc(Assembler::zero, FALSE_LABEL);
4332 testptr(ary2, ary2);
4333 jcc(Assembler::zero, FALSE_LABEL);
4334
4335 // Check the lengths
4336 movl(limit, Address(ary1, length_offset));
4337 cmpl(limit, Address(ary2, length_offset));
4338 jcc(Assembler::notEqual, FALSE_LABEL);
4339 }
4340
4341 // count == 0
4342 testl(limit, limit);
4343 jcc(Assembler::zero, TRUE_LABEL);
4344
4345 if (is_array_equ) {
4346 // Load array address
4347 lea(ary1, Address(ary1, base_offset));
4348 lea(ary2, Address(ary2, base_offset));
4349 }
4350
4351 if (is_array_equ && is_char) {
4352 // arrays_equals when used for char[].
4353 shll(limit, 1); // byte count != 0
4354 }
4355 movl(result, limit); // copy
4356
4357 if (UseAVX >= 2) {
4358 // With AVX2, use 32-byte vector compare
4359 Label COMPARE_WIDE_VECTORS, COMPARE_WIDE_VECTORS_16, COMPARE_TAIL, COMPARE_TAIL_16;
4360
4361 // Compare 32-byte vectors
4362 if (expand_ary2) {
4363 andl(result, 0x0000000f); // tail count (in bytes)
4364 andl(limit, 0xfffffff0); // vector count (in bytes)
4365 jcc(Assembler::zero, COMPARE_TAIL);
4366 } else {
4367 andl(result, 0x0000001f); // tail count (in bytes)
4368 andl(limit, 0xffffffe0); // vector count (in bytes)
4369 jcc(Assembler::zero, COMPARE_TAIL_16);
4370 }
4371
4372 lea(ary1, Address(ary1, limit, scaleFactor));
4373 lea(ary2, Address(ary2, limit, Address::times_1));
4374 negptr(limit);
4375
4376 if ((AVX3Threshold == 0) && VM_Version::supports_avx512vlbw()) { // trying 64 bytes fast loop
4377 Label COMPARE_WIDE_VECTORS_LOOP_AVX2, COMPARE_WIDE_VECTORS_LOOP_AVX3;
4378
4379 cmpl(limit, -64);
4380 jcc(Assembler::greater, COMPARE_WIDE_VECTORS_LOOP_AVX2);
4381
4382 bind(COMPARE_WIDE_VECTORS_LOOP_AVX3); // the hottest loop
4383
4384 evmovdquq(vec1, Address(ary1, limit, Address::times_1), Assembler::AVX_512bit);
4385 evpcmpeqb(mask, vec1, Address(ary2, limit, Address::times_1), Assembler::AVX_512bit);
4386 kortestql(mask, mask);
4387 jcc(Assembler::aboveEqual, FALSE_LABEL); // miscompare
4388 addptr(limit, 64); // update since we already compared at this addr
4389 cmpl(limit, -64);
4390 jccb(Assembler::lessEqual, COMPARE_WIDE_VECTORS_LOOP_AVX3);
4391
4392 // At this point we may still need to compare -limit+result bytes.
4393 // We could execute the next two instruction and just continue via non-wide path:
4394 // cmpl(limit, 0);
4395 // jcc(Assembler::equal, COMPARE_TAIL); // true
4396 // But since we stopped at the points ary{1,2}+limit which are
4397 // not farther than 64 bytes from the ends of arrays ary{1,2}+result
4398 // (|limit| <= 32 and result < 32),
4399 // we may just compare the last 64 bytes.
4400 //
4401 addptr(result, -64); // it is safe, bc we just came from this area
4402 evmovdquq(vec1, Address(ary1, result, Address::times_1), Assembler::AVX_512bit);
4403 evpcmpeqb(mask, vec1, Address(ary2, result, Address::times_1), Assembler::AVX_512bit);
4404 kortestql(mask, mask);
4405 jcc(Assembler::aboveEqual, FALSE_LABEL); // miscompare
4406
4407 jmp(TRUE_LABEL);
4408
4409 bind(COMPARE_WIDE_VECTORS_LOOP_AVX2);
4410
4411 }//if (VM_Version::supports_avx512vlbw())
4412
4413 bind(COMPARE_WIDE_VECTORS);
4414 vmovdqu(vec1, Address(ary1, limit, scaleFactor));
4415 if (expand_ary2) {
4416 vpmovzxbw(vec2, Address(ary2, limit, Address::times_1), Assembler::AVX_256bit);
4417 } else {
4418 vmovdqu(vec2, Address(ary2, limit, Address::times_1));
4419 }
4420 vpxor(vec1, vec2);
4421
4422 vptest(vec1, vec1);
4423 jcc(Assembler::notZero, FALSE_LABEL);
4424 addptr(limit, scaleIncr * 2);
4425 jcc(Assembler::notZero, COMPARE_WIDE_VECTORS);
4426
4427 testl(result, result);
4428 jcc(Assembler::zero, TRUE_LABEL);
4429
4430 vmovdqu(vec1, Address(ary1, result, scaleFactor, -32));
4431 if (expand_ary2) {
4432 vpmovzxbw(vec2, Address(ary2, result, Address::times_1, -16), Assembler::AVX_256bit);
4433 } else {
4434 vmovdqu(vec2, Address(ary2, result, Address::times_1, -32));
4435 }
4436 vpxor(vec1, vec2);
4437
4438 vptest(vec1, vec1);
4439 jcc(Assembler::notZero, FALSE_LABEL);
4440 jmp(TRUE_LABEL);
4441
4442 bind(COMPARE_TAIL_16); // limit is zero
4443 movl(limit, result);
4444
4445 // Compare 16-byte chunks
4446 andl(result, 0x0000000f); // tail count (in bytes)
4447 andl(limit, 0xfffffff0); // vector count (in bytes)
4448 jcc(Assembler::zero, COMPARE_TAIL);
4449
4450 lea(ary1, Address(ary1, limit, scaleFactor));
4451 lea(ary2, Address(ary2, limit, Address::times_1));
4452 negptr(limit);
4453
4454 bind(COMPARE_WIDE_VECTORS_16);
4455 movdqu(vec1, Address(ary1, limit, scaleFactor));
4456 if (expand_ary2) {
4457 vpmovzxbw(vec2, Address(ary2, limit, Address::times_1), Assembler::AVX_128bit);
4458 } else {
4459 movdqu(vec2, Address(ary2, limit, Address::times_1));
4460 }
4461 pxor(vec1, vec2);
4462
4463 ptest(vec1, vec1);
4464 jcc(Assembler::notZero, FALSE_LABEL);
4465 addptr(limit, scaleIncr);
4466 jcc(Assembler::notZero, COMPARE_WIDE_VECTORS_16);
4467
4468 bind(COMPARE_TAIL); // limit is zero
4469 movl(limit, result);
4470 // Fallthru to tail compare
4471 } else if (UseSSE42Intrinsics) {
4472 // With SSE4.2, use double quad vector compare
4473 Label COMPARE_WIDE_VECTORS, COMPARE_TAIL;
4474
4475 // Compare 16-byte vectors
4476 andl(result, 0x0000000f); // tail count (in bytes)
4477 andl(limit, 0xfffffff0); // vector count (in bytes)
4478 jcc(Assembler::zero, COMPARE_TAIL);
4479
4480 lea(ary1, Address(ary1, limit, Address::times_1));
4481 lea(ary2, Address(ary2, limit, Address::times_1));
4482 negptr(limit);
4483
4484 bind(COMPARE_WIDE_VECTORS);
4485 movdqu(vec1, Address(ary1, limit, Address::times_1));
4486 movdqu(vec2, Address(ary2, limit, Address::times_1));
4487 pxor(vec1, vec2);
4488
4489 ptest(vec1, vec1);
4490 jcc(Assembler::notZero, FALSE_LABEL);
4491 addptr(limit, 16);
4492 jcc(Assembler::notZero, COMPARE_WIDE_VECTORS);
4493
4494 testl(result, result);
4495 jcc(Assembler::zero, TRUE_LABEL);
4496
4497 movdqu(vec1, Address(ary1, result, Address::times_1, -16));
4498 movdqu(vec2, Address(ary2, result, Address::times_1, -16));
4499 pxor(vec1, vec2);
4500
4501 ptest(vec1, vec1);
4502 jccb(Assembler::notZero, FALSE_LABEL);
4503 jmpb(TRUE_LABEL);
4504
4505 bind(COMPARE_TAIL); // limit is zero
4506 movl(limit, result);
4507 // Fallthru to tail compare
4508 }
4509
4510 // Compare 4-byte vectors
4511 if (expand_ary2) {
4512 testl(result, result);
4513 jccb(Assembler::zero, TRUE_LABEL);
4514 } else {
4515 andl(limit, 0xfffffffc); // vector count (in bytes)
4516 jccb(Assembler::zero, COMPARE_CHAR);
4517 }
4518
4519 lea(ary1, Address(ary1, limit, scaleFactor));
4520 lea(ary2, Address(ary2, limit, Address::times_1));
4521 negptr(limit);
4522
4523 bind(COMPARE_VECTORS);
4524 if (expand_ary2) {
4525 // There are no "vector" operations for bytes to shorts
4526 movzbl(chr, Address(ary2, limit, Address::times_1));
4527 cmpw(Address(ary1, limit, Address::times_2), chr);
4528 jccb(Assembler::notEqual, FALSE_LABEL);
4529 addptr(limit, 1);
4530 jcc(Assembler::notZero, COMPARE_VECTORS);
4531 jmp(TRUE_LABEL);
4532 } else {
4533 movl(chr, Address(ary1, limit, Address::times_1));
4534 cmpl(chr, Address(ary2, limit, Address::times_1));
4535 jccb(Assembler::notEqual, FALSE_LABEL);
4536 addptr(limit, 4);
4537 jcc(Assembler::notZero, COMPARE_VECTORS);
4538 }
4539
4540 // Compare trailing char (final 2 bytes), if any
4541 bind(COMPARE_CHAR);
4542 testl(result, 0x2); // tail char
4543 jccb(Assembler::zero, COMPARE_BYTE);
4544 load_unsigned_short(chr, Address(ary1, 0));
4545 load_unsigned_short(limit, Address(ary2, 0));
4546 cmpl(chr, limit);
4547 jccb(Assembler::notEqual, FALSE_LABEL);
4548
4549 if (is_array_equ && is_char) {
4550 bind(COMPARE_BYTE);
4551 } else {
4552 lea(ary1, Address(ary1, 2));
4553 lea(ary2, Address(ary2, 2));
4554
4555 bind(COMPARE_BYTE);
4556 testl(result, 0x1); // tail byte
4557 jccb(Assembler::zero, TRUE_LABEL);
4558 load_unsigned_byte(chr, Address(ary1, 0));
4559 load_unsigned_byte(limit, Address(ary2, 0));
4560 cmpl(chr, limit);
4561 jccb(Assembler::notEqual, FALSE_LABEL);
4562 }
4563 bind(TRUE_LABEL);
4564 movl(result, 1); // return true
4565 jmpb(DONE);
4566
4567 bind(FALSE_LABEL);
4568 xorl(result, result); // return false
4569
4570 // That's it
4571 bind(DONE);
4572 if (UseAVX >= 2) {
4573 // clean upper bits of YMM registers
4574 vpxor(vec1, vec1);
4575 vpxor(vec2, vec2);
4576 }
4577 }
4578
4579 static void convertF2I_slowpath(C2_MacroAssembler& masm, C2GeneralStub<Register, XMMRegister, address>& stub) {
4580 #define __ masm.
4581 Register dst = stub.data<0>();
4582 XMMRegister src = stub.data<1>();
4583 address target = stub.data<2>();
4584 __ bind(stub.entry());
4585 __ subptr(rsp, 8);
4586 __ movdbl(Address(rsp), src);
4587 __ call(RuntimeAddress(target));
4588 // APX REX2 encoding for pop(dst) increases the stub size by 1 byte.
4589 __ pop(dst);
4590 __ jmp(stub.continuation());
4591 #undef __
4592 }
4593
4594 void C2_MacroAssembler::convertF2I(BasicType dst_bt, BasicType src_bt, Register dst, XMMRegister src) {
4595 assert(dst_bt == T_INT || dst_bt == T_LONG, "");
4596 assert(src_bt == T_FLOAT || src_bt == T_DOUBLE, "");
4597
4598 address slowpath_target;
4599 if (dst_bt == T_INT) {
4600 if (src_bt == T_FLOAT) {
4601 cvttss2sil(dst, src);
4602 cmpl(dst, 0x80000000);
4603 slowpath_target = StubRoutines::x86::f2i_fixup();
4604 } else {
4605 cvttsd2sil(dst, src);
4606 cmpl(dst, 0x80000000);
4607 slowpath_target = StubRoutines::x86::d2i_fixup();
4608 }
4609 } else {
4610 if (src_bt == T_FLOAT) {
4611 cvttss2siq(dst, src);
4612 cmp64(dst, ExternalAddress(StubRoutines::x86::double_sign_flip()));
4613 slowpath_target = StubRoutines::x86::f2l_fixup();
4614 } else {
4615 cvttsd2siq(dst, src);
4616 cmp64(dst, ExternalAddress(StubRoutines::x86::double_sign_flip()));
4617 slowpath_target = StubRoutines::x86::d2l_fixup();
4618 }
4619 }
4620
4621 // Using the APX extended general purpose registers increases the instruction encoding size by 1 byte.
4622 int max_size = 23 + (UseAPX ? 1 : 0);
4623 auto stub = C2CodeStub::make<Register, XMMRegister, address>(dst, src, slowpath_target, max_size, convertF2I_slowpath);
4624 jcc(Assembler::equal, stub->entry());
4625 bind(stub->continuation());
4626 }
4627
4628 void C2_MacroAssembler::evmasked_op(int ideal_opc, BasicType eType, KRegister mask, XMMRegister dst,
4629 XMMRegister src1, int imm8, bool merge, int vlen_enc) {
4630 switch(ideal_opc) {
4631 case Op_LShiftVS:
4632 Assembler::evpsllw(dst, mask, src1, imm8, merge, vlen_enc); break;
4633 case Op_LShiftVI:
4634 Assembler::evpslld(dst, mask, src1, imm8, merge, vlen_enc); break;
4635 case Op_LShiftVL:
4636 Assembler::evpsllq(dst, mask, src1, imm8, merge, vlen_enc); break;
4637 case Op_RShiftVS:
4638 Assembler::evpsraw(dst, mask, src1, imm8, merge, vlen_enc); break;
4639 case Op_RShiftVI:
4640 Assembler::evpsrad(dst, mask, src1, imm8, merge, vlen_enc); break;
4641 case Op_RShiftVL:
4642 Assembler::evpsraq(dst, mask, src1, imm8, merge, vlen_enc); break;
4643 case Op_URShiftVS:
4644 Assembler::evpsrlw(dst, mask, src1, imm8, merge, vlen_enc); break;
4645 case Op_URShiftVI:
4646 Assembler::evpsrld(dst, mask, src1, imm8, merge, vlen_enc); break;
4647 case Op_URShiftVL:
4648 Assembler::evpsrlq(dst, mask, src1, imm8, merge, vlen_enc); break;
4649 case Op_RotateRightV:
4650 evrord(eType, dst, mask, src1, imm8, merge, vlen_enc); break;
4651 case Op_RotateLeftV:
4652 evrold(eType, dst, mask, src1, imm8, merge, vlen_enc); break;
4653 default:
4654 fatal("Unsupported operation %s", NodeClassNames[ideal_opc]);
4655 break;
4656 }
4657 }
4658
4659 void C2_MacroAssembler::evmasked_saturating_op(int ideal_opc, BasicType elem_bt, KRegister mask, XMMRegister dst, XMMRegister src1,
4660 XMMRegister src2, bool is_unsigned, bool merge, int vlen_enc) {
4661 if (is_unsigned) {
4662 evmasked_saturating_unsigned_op(ideal_opc, elem_bt, mask, dst, src1, src2, merge, vlen_enc);
4663 } else {
4664 evmasked_saturating_signed_op(ideal_opc, elem_bt, mask, dst, src1, src2, merge, vlen_enc);
4665 }
4666 }
4667
4668 void C2_MacroAssembler::evmasked_saturating_signed_op(int ideal_opc, BasicType elem_bt, KRegister mask, XMMRegister dst,
4669 XMMRegister src1, XMMRegister src2, bool merge, int vlen_enc) {
4670 switch (elem_bt) {
4671 case T_BYTE:
4672 if (ideal_opc == Op_SaturatingAddV) {
4673 evpaddsb(dst, mask, src1, src2, merge, vlen_enc);
4674 } else {
4675 assert(ideal_opc == Op_SaturatingSubV, "");
4676 evpsubsb(dst, mask, src1, src2, merge, vlen_enc);
4677 }
4678 break;
4679 case T_SHORT:
4680 if (ideal_opc == Op_SaturatingAddV) {
4681 evpaddsw(dst, mask, src1, src2, merge, vlen_enc);
4682 } else {
4683 assert(ideal_opc == Op_SaturatingSubV, "");
4684 evpsubsw(dst, mask, src1, src2, merge, vlen_enc);
4685 }
4686 break;
4687 default:
4688 fatal("Unsupported type %s", type2name(elem_bt));
4689 break;
4690 }
4691 }
4692
4693 void C2_MacroAssembler::evmasked_saturating_unsigned_op(int ideal_opc, BasicType elem_bt, KRegister mask, XMMRegister dst,
4694 XMMRegister src1, XMMRegister src2, bool merge, int vlen_enc) {
4695 switch (elem_bt) {
4696 case T_BYTE:
4697 if (ideal_opc == Op_SaturatingAddV) {
4698 evpaddusb(dst, mask, src1, src2, merge, vlen_enc);
4699 } else {
4700 assert(ideal_opc == Op_SaturatingSubV, "");
4701 evpsubusb(dst, mask, src1, src2, merge, vlen_enc);
4702 }
4703 break;
4704 case T_SHORT:
4705 if (ideal_opc == Op_SaturatingAddV) {
4706 evpaddusw(dst, mask, src1, src2, merge, vlen_enc);
4707 } else {
4708 assert(ideal_opc == Op_SaturatingSubV, "");
4709 evpsubusw(dst, mask, src1, src2, merge, vlen_enc);
4710 }
4711 break;
4712 default:
4713 fatal("Unsupported type %s", type2name(elem_bt));
4714 break;
4715 }
4716 }
4717
4718 void C2_MacroAssembler::evmasked_saturating_op(int ideal_opc, BasicType elem_bt, KRegister mask, XMMRegister dst, XMMRegister src1,
4719 Address src2, bool is_unsigned, bool merge, int vlen_enc) {
4720 if (is_unsigned) {
4721 evmasked_saturating_unsigned_op(ideal_opc, elem_bt, mask, dst, src1, src2, merge, vlen_enc);
4722 } else {
4723 evmasked_saturating_signed_op(ideal_opc, elem_bt, mask, dst, src1, src2, merge, vlen_enc);
4724 }
4725 }
4726
4727 void C2_MacroAssembler::evmasked_saturating_signed_op(int ideal_opc, BasicType elem_bt, KRegister mask, XMMRegister dst,
4728 XMMRegister src1, Address src2, bool merge, int vlen_enc) {
4729 switch (elem_bt) {
4730 case T_BYTE:
4731 if (ideal_opc == Op_SaturatingAddV) {
4732 evpaddsb(dst, mask, src1, src2, merge, vlen_enc);
4733 } else {
4734 assert(ideal_opc == Op_SaturatingSubV, "");
4735 evpsubsb(dst, mask, src1, src2, merge, vlen_enc);
4736 }
4737 break;
4738 case T_SHORT:
4739 if (ideal_opc == Op_SaturatingAddV) {
4740 evpaddsw(dst, mask, src1, src2, merge, vlen_enc);
4741 } else {
4742 assert(ideal_opc == Op_SaturatingSubV, "");
4743 evpsubsw(dst, mask, src1, src2, merge, vlen_enc);
4744 }
4745 break;
4746 default:
4747 fatal("Unsupported type %s", type2name(elem_bt));
4748 break;
4749 }
4750 }
4751
4752 void C2_MacroAssembler::evmasked_saturating_unsigned_op(int ideal_opc, BasicType elem_bt, KRegister mask, XMMRegister dst,
4753 XMMRegister src1, Address src2, bool merge, int vlen_enc) {
4754 switch (elem_bt) {
4755 case T_BYTE:
4756 if (ideal_opc == Op_SaturatingAddV) {
4757 evpaddusb(dst, mask, src1, src2, merge, vlen_enc);
4758 } else {
4759 assert(ideal_opc == Op_SaturatingSubV, "");
4760 evpsubusb(dst, mask, src1, src2, merge, vlen_enc);
4761 }
4762 break;
4763 case T_SHORT:
4764 if (ideal_opc == Op_SaturatingAddV) {
4765 evpaddusw(dst, mask, src1, src2, merge, vlen_enc);
4766 } else {
4767 assert(ideal_opc == Op_SaturatingSubV, "");
4768 evpsubusw(dst, mask, src1, src2, merge, vlen_enc);
4769 }
4770 break;
4771 default:
4772 fatal("Unsupported type %s", type2name(elem_bt));
4773 break;
4774 }
4775 }
4776
4777 void C2_MacroAssembler::evmasked_op(int ideal_opc, BasicType eType, KRegister mask, XMMRegister dst,
4778 XMMRegister src1, XMMRegister src2, bool merge, int vlen_enc,
4779 bool is_varshift) {
4780 switch (ideal_opc) {
4781 case Op_AddVB:
4782 evpaddb(dst, mask, src1, src2, merge, vlen_enc); break;
4783 case Op_AddVS:
4784 evpaddw(dst, mask, src1, src2, merge, vlen_enc); break;
4785 case Op_AddVI:
4786 evpaddd(dst, mask, src1, src2, merge, vlen_enc); break;
4787 case Op_AddVL:
4788 evpaddq(dst, mask, src1, src2, merge, vlen_enc); break;
4789 case Op_AddVF:
4790 evaddps(dst, mask, src1, src2, merge, vlen_enc); break;
4791 case Op_AddVD:
4792 evaddpd(dst, mask, src1, src2, merge, vlen_enc); break;
4793 case Op_SubVB:
4794 evpsubb(dst, mask, src1, src2, merge, vlen_enc); break;
4795 case Op_SubVS:
4796 evpsubw(dst, mask, src1, src2, merge, vlen_enc); break;
4797 case Op_SubVI:
4798 evpsubd(dst, mask, src1, src2, merge, vlen_enc); break;
4799 case Op_SubVL:
4800 evpsubq(dst, mask, src1, src2, merge, vlen_enc); break;
4801 case Op_SubVF:
4802 evsubps(dst, mask, src1, src2, merge, vlen_enc); break;
4803 case Op_SubVD:
4804 evsubpd(dst, mask, src1, src2, merge, vlen_enc); break;
4805 case Op_MulVS:
4806 evpmullw(dst, mask, src1, src2, merge, vlen_enc); break;
4807 case Op_MulVI:
4808 evpmulld(dst, mask, src1, src2, merge, vlen_enc); break;
4809 case Op_MulVL:
4810 evpmullq(dst, mask, src1, src2, merge, vlen_enc); break;
4811 case Op_MulVF:
4812 evmulps(dst, mask, src1, src2, merge, vlen_enc); break;
4813 case Op_MulVD:
4814 evmulpd(dst, mask, src1, src2, merge, vlen_enc); break;
4815 case Op_DivVF:
4816 evdivps(dst, mask, src1, src2, merge, vlen_enc); break;
4817 case Op_DivVD:
4818 evdivpd(dst, mask, src1, src2, merge, vlen_enc); break;
4819 case Op_SqrtVF:
4820 evsqrtps(dst, mask, src1, src2, merge, vlen_enc); break;
4821 case Op_SqrtVD:
4822 evsqrtpd(dst, mask, src1, src2, merge, vlen_enc); break;
4823 case Op_AbsVB:
4824 evpabsb(dst, mask, src2, merge, vlen_enc); break;
4825 case Op_AbsVS:
4826 evpabsw(dst, mask, src2, merge, vlen_enc); break;
4827 case Op_AbsVI:
4828 evpabsd(dst, mask, src2, merge, vlen_enc); break;
4829 case Op_AbsVL:
4830 evpabsq(dst, mask, src2, merge, vlen_enc); break;
4831 case Op_FmaVF:
4832 evpfma213ps(dst, mask, src1, src2, merge, vlen_enc); break;
4833 case Op_FmaVD:
4834 evpfma213pd(dst, mask, src1, src2, merge, vlen_enc); break;
4835 case Op_VectorRearrange:
4836 evperm(eType, dst, mask, src2, src1, merge, vlen_enc); break;
4837 case Op_LShiftVS:
4838 evpsllw(dst, mask, src1, src2, merge, vlen_enc, is_varshift); break;
4839 case Op_LShiftVI:
4840 evpslld(dst, mask, src1, src2, merge, vlen_enc, is_varshift); break;
4841 case Op_LShiftVL:
4842 evpsllq(dst, mask, src1, src2, merge, vlen_enc, is_varshift); break;
4843 case Op_RShiftVS:
4844 evpsraw(dst, mask, src1, src2, merge, vlen_enc, is_varshift); break;
4845 case Op_RShiftVI:
4846 evpsrad(dst, mask, src1, src2, merge, vlen_enc, is_varshift); break;
4847 case Op_RShiftVL:
4848 evpsraq(dst, mask, src1, src2, merge, vlen_enc, is_varshift); break;
4849 case Op_URShiftVS:
4850 evpsrlw(dst, mask, src1, src2, merge, vlen_enc, is_varshift); break;
4851 case Op_URShiftVI:
4852 evpsrld(dst, mask, src1, src2, merge, vlen_enc, is_varshift); break;
4853 case Op_URShiftVL:
4854 evpsrlq(dst, mask, src1, src2, merge, vlen_enc, is_varshift); break;
4855 case Op_RotateLeftV:
4856 evrold(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4857 case Op_RotateRightV:
4858 evrord(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4859 case Op_MaxV:
4860 evpmaxs(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4861 case Op_MinV:
4862 evpmins(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4863 case Op_UMinV:
4864 evpminu(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4865 case Op_UMaxV:
4866 evpmaxu(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4867 case Op_XorV:
4868 evxor(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4869 case Op_OrV:
4870 evor(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4871 case Op_AndV:
4872 evand(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4873 default:
4874 fatal("Unsupported operation %s", NodeClassNames[ideal_opc]);
4875 break;
4876 }
4877 }
4878
4879 void C2_MacroAssembler::evmasked_op(int ideal_opc, BasicType eType, KRegister mask, XMMRegister dst,
4880 XMMRegister src1, Address src2, bool merge, int vlen_enc) {
4881 switch (ideal_opc) {
4882 case Op_AddVB:
4883 evpaddb(dst, mask, src1, src2, merge, vlen_enc); break;
4884 case Op_AddVS:
4885 evpaddw(dst, mask, src1, src2, merge, vlen_enc); break;
4886 case Op_AddVI:
4887 evpaddd(dst, mask, src1, src2, merge, vlen_enc); break;
4888 case Op_AddVL:
4889 evpaddq(dst, mask, src1, src2, merge, vlen_enc); break;
4890 case Op_AddVF:
4891 evaddps(dst, mask, src1, src2, merge, vlen_enc); break;
4892 case Op_AddVD:
4893 evaddpd(dst, mask, src1, src2, merge, vlen_enc); break;
4894 case Op_SubVB:
4895 evpsubb(dst, mask, src1, src2, merge, vlen_enc); break;
4896 case Op_SubVS:
4897 evpsubw(dst, mask, src1, src2, merge, vlen_enc); break;
4898 case Op_SubVI:
4899 evpsubd(dst, mask, src1, src2, merge, vlen_enc); break;
4900 case Op_SubVL:
4901 evpsubq(dst, mask, src1, src2, merge, vlen_enc); break;
4902 case Op_SubVF:
4903 evsubps(dst, mask, src1, src2, merge, vlen_enc); break;
4904 case Op_SubVD:
4905 evsubpd(dst, mask, src1, src2, merge, vlen_enc); break;
4906 case Op_MulVS:
4907 evpmullw(dst, mask, src1, src2, merge, vlen_enc); break;
4908 case Op_MulVI:
4909 evpmulld(dst, mask, src1, src2, merge, vlen_enc); break;
4910 case Op_MulVL:
4911 evpmullq(dst, mask, src1, src2, merge, vlen_enc); break;
4912 case Op_MulVF:
4913 evmulps(dst, mask, src1, src2, merge, vlen_enc); break;
4914 case Op_MulVD:
4915 evmulpd(dst, mask, src1, src2, merge, vlen_enc); break;
4916 case Op_DivVF:
4917 evdivps(dst, mask, src1, src2, merge, vlen_enc); break;
4918 case Op_DivVD:
4919 evdivpd(dst, mask, src1, src2, merge, vlen_enc); break;
4920 case Op_FmaVF:
4921 evpfma213ps(dst, mask, src1, src2, merge, vlen_enc); break;
4922 case Op_FmaVD:
4923 evpfma213pd(dst, mask, src1, src2, merge, vlen_enc); break;
4924 case Op_MaxV:
4925 evpmaxs(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4926 case Op_MinV:
4927 evpmins(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4928 case Op_UMaxV:
4929 evpmaxu(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4930 case Op_UMinV:
4931 evpminu(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4932 case Op_XorV:
4933 evxor(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4934 case Op_OrV:
4935 evor(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4936 case Op_AndV:
4937 evand(eType, dst, mask, src1, src2, merge, vlen_enc); break;
4938 default:
4939 fatal("Unsupported operation %s", NodeClassNames[ideal_opc]);
4940 break;
4941 }
4942 }
4943
4944 void C2_MacroAssembler::masked_op(int ideal_opc, int mask_len, KRegister dst,
4945 KRegister src1, KRegister src2) {
4946 BasicType etype = T_ILLEGAL;
4947 switch(mask_len) {
4948 case 2:
4949 case 4:
4950 case 8: etype = T_BYTE; break;
4951 case 16: etype = T_SHORT; break;
4952 case 32: etype = T_INT; break;
4953 case 64: etype = T_LONG; break;
4954 default: fatal("Unsupported type"); break;
4955 }
4956 assert(etype != T_ILLEGAL, "");
4957 switch(ideal_opc) {
4958 case Op_AndVMask:
4959 kand(etype, dst, src1, src2); break;
4960 case Op_OrVMask:
4961 kor(etype, dst, src1, src2); break;
4962 case Op_XorVMask:
4963 kxor(etype, dst, src1, src2); break;
4964 default:
4965 fatal("Unsupported masked operation"); break;
4966 }
4967 }
4968
4969 /*
4970 * Following routine handles special floating point values(NaN/Inf/-Inf/Max/Min) for casting operation.
4971 * If src is NaN, the result is 0.
4972 * If the src is negative infinity or any value less than or equal to the value of Integer.MIN_VALUE,
4973 * the result is equal to the value of Integer.MIN_VALUE.
4974 * If the src is positive infinity or any value greater than or equal to the value of Integer.MAX_VALUE,
4975 * the result is equal to the value of Integer.MAX_VALUE.
4976 */
4977 void C2_MacroAssembler::vector_cast_float_to_int_special_cases_avx(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
4978 XMMRegister xtmp2, XMMRegister xtmp3, XMMRegister xtmp4,
4979 Register rscratch, AddressLiteral float_sign_flip,
4980 int vec_enc) {
4981 assert(rscratch != noreg || always_reachable(float_sign_flip), "missing");
4982 Label done;
4983 vmovdqu(xtmp1, float_sign_flip, vec_enc, rscratch);
4984 vpcmpeqd(xtmp2, dst, xtmp1, vec_enc);
4985 vptest(xtmp2, xtmp2, vec_enc);
4986 jccb(Assembler::equal, done);
4987
4988 vpcmpeqd(xtmp4, xtmp4, xtmp4, vec_enc);
4989 vpxor(xtmp1, xtmp1, xtmp4, vec_enc);
4990
4991 vpxor(xtmp4, xtmp4, xtmp4, vec_enc);
4992 vcmpps(xtmp3, src, src, Assembler::UNORD_Q, vec_enc);
4993 vblendvps(dst, dst, xtmp4, xtmp3, vec_enc);
4994
4995 // Recompute the mask for remaining special value.
4996 vpxor(xtmp2, xtmp2, xtmp3, vec_enc);
4997 // Extract SRC values corresponding to TRUE mask lanes.
4998 vpand(xtmp4, xtmp2, src, vec_enc);
4999 // Flip mask bits so that MSB bit of MASK lanes corresponding to +ve special
5000 // values are set.
5001 vpxor(xtmp3, xtmp2, xtmp4, vec_enc);
5002
5003 vblendvps(dst, dst, xtmp1, xtmp3, vec_enc);
5004 bind(done);
5005 }
5006
5007 void C2_MacroAssembler::vector_cast_float_to_int_special_cases_evex(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5008 XMMRegister xtmp2, KRegister ktmp1, KRegister ktmp2,
5009 Register rscratch, AddressLiteral float_sign_flip,
5010 int vec_enc) {
5011 assert(rscratch != noreg || always_reachable(float_sign_flip), "missing");
5012 Label done;
5013 evmovdqul(xtmp1, k0, float_sign_flip, false, vec_enc, rscratch);
5014 Assembler::evpcmpeqd(ktmp1, k0, xtmp1, dst, vec_enc);
5015 kortestwl(ktmp1, ktmp1);
5016 jccb(Assembler::equal, done);
5017
5018 vpxor(xtmp2, xtmp2, xtmp2, vec_enc);
5019 evcmpps(ktmp2, k0, src, src, Assembler::UNORD_Q, vec_enc);
5020 evmovdqul(dst, ktmp2, xtmp2, true, vec_enc);
5021
5022 kxorwl(ktmp1, ktmp1, ktmp2);
5023 evcmpps(ktmp1, ktmp1, src, xtmp2, Assembler::NLT_UQ, vec_enc);
5024 vpternlogd(xtmp2, 0x11, xtmp1, xtmp1, vec_enc);
5025 evmovdqul(dst, ktmp1, xtmp2, true, vec_enc);
5026 bind(done);
5027 }
5028
5029 void C2_MacroAssembler::vector_cast_float_to_long_special_cases_evex(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5030 XMMRegister xtmp2, KRegister ktmp1, KRegister ktmp2,
5031 Register rscratch, AddressLiteral double_sign_flip,
5032 int vec_enc) {
5033 assert(rscratch != noreg || always_reachable(double_sign_flip), "missing");
5034
5035 Label done;
5036 evmovdquq(xtmp1, k0, double_sign_flip, false, vec_enc, rscratch);
5037 Assembler::evpcmpeqq(ktmp1, k0, xtmp1, dst, vec_enc);
5038 kortestwl(ktmp1, ktmp1);
5039 jccb(Assembler::equal, done);
5040
5041 vpxor(xtmp2, xtmp2, xtmp2, vec_enc);
5042 evcmpps(ktmp2, k0, src, src, Assembler::UNORD_Q, vec_enc);
5043 evmovdquq(dst, ktmp2, xtmp2, true, vec_enc);
5044
5045 kxorwl(ktmp1, ktmp1, ktmp2);
5046 evcmpps(ktmp1, ktmp1, src, xtmp2, Assembler::NLT_UQ, vec_enc);
5047 vpternlogq(xtmp2, 0x11, xtmp1, xtmp1, vec_enc);
5048 evmovdquq(dst, ktmp1, xtmp2, true, vec_enc);
5049 bind(done);
5050 }
5051
5052 void C2_MacroAssembler::vector_cast_double_to_int_special_cases_evex(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5053 XMMRegister xtmp2, KRegister ktmp1, KRegister ktmp2,
5054 Register rscratch, AddressLiteral float_sign_flip,
5055 int vec_enc) {
5056 assert(rscratch != noreg || always_reachable(float_sign_flip), "missing");
5057 Label done;
5058 evmovdquq(xtmp1, k0, float_sign_flip, false, vec_enc, rscratch);
5059 Assembler::evpcmpeqd(ktmp1, k0, xtmp1, dst, vec_enc);
5060 kortestwl(ktmp1, ktmp1);
5061 jccb(Assembler::equal, done);
5062
5063 vpxor(xtmp2, xtmp2, xtmp2, vec_enc);
5064 evcmppd(ktmp2, k0, src, src, Assembler::UNORD_Q, vec_enc);
5065 evmovdqul(dst, ktmp2, xtmp2, true, vec_enc);
5066
5067 kxorwl(ktmp1, ktmp1, ktmp2);
5068 evcmppd(ktmp1, ktmp1, src, xtmp2, Assembler::NLT_UQ, vec_enc);
5069 vpternlogq(xtmp2, 0x11, xtmp1, xtmp1, vec_enc);
5070 evmovdqul(dst, ktmp1, xtmp2, true, vec_enc);
5071 bind(done);
5072 }
5073
5074 /*
5075 * Following routine handles special floating point values(NaN/Inf/-Inf/Max/Min) for casting operation.
5076 * If src is NaN, the result is 0.
5077 * If the src is negative infinity or any value less than or equal to the value of Long.MIN_VALUE,
5078 * the result is equal to the value of Long.MIN_VALUE.
5079 * If the src is positive infinity or any value greater than or equal to the value of Long.MAX_VALUE,
5080 * the result is equal to the value of Long.MAX_VALUE.
5081 */
5082 void C2_MacroAssembler::vector_cast_double_to_long_special_cases_evex(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5083 XMMRegister xtmp2, KRegister ktmp1, KRegister ktmp2,
5084 Register rscratch, AddressLiteral double_sign_flip,
5085 int vec_enc) {
5086 assert(rscratch != noreg || always_reachable(double_sign_flip), "missing");
5087
5088 Label done;
5089 evmovdqul(xtmp1, k0, double_sign_flip, false, vec_enc, rscratch);
5090 evpcmpeqq(ktmp1, xtmp1, dst, vec_enc);
5091 kortestwl(ktmp1, ktmp1);
5092 jccb(Assembler::equal, done);
5093
5094 vpxor(xtmp2, xtmp2, xtmp2, vec_enc);
5095 evcmppd(ktmp2, k0, src, src, Assembler::UNORD_Q, vec_enc);
5096 evmovdquq(dst, ktmp2, xtmp2, true, vec_enc);
5097
5098 kxorwl(ktmp1, ktmp1, ktmp2);
5099 evcmppd(ktmp1, ktmp1, src, xtmp2, Assembler::NLT_UQ, vec_enc);
5100 vpternlogq(xtmp2, 0x11, xtmp1, xtmp1, vec_enc);
5101 evmovdquq(dst, ktmp1, xtmp2, true, vec_enc);
5102 bind(done);
5103 }
5104
5105 void C2_MacroAssembler::vector_crosslane_doubleword_pack_avx(XMMRegister dst, XMMRegister src, XMMRegister zero,
5106 XMMRegister xtmp, int index, int vec_enc) {
5107 assert(vec_enc < Assembler::AVX_512bit, "");
5108 if (vec_enc == Assembler::AVX_256bit) {
5109 vextractf128_high(xtmp, src);
5110 vshufps(dst, src, xtmp, index, vec_enc);
5111 } else {
5112 vshufps(dst, src, zero, index, vec_enc);
5113 }
5114 }
5115
5116 void C2_MacroAssembler::vector_cast_double_to_int_special_cases_avx(XMMRegister dst, XMMRegister src, XMMRegister xtmp1, XMMRegister xtmp2,
5117 XMMRegister xtmp3, XMMRegister xtmp4, XMMRegister xtmp5, Register rscratch,
5118 AddressLiteral float_sign_flip, int src_vec_enc) {
5119 assert(rscratch != noreg || always_reachable(float_sign_flip), "missing");
5120
5121 Label done;
5122 // Compare the destination lanes with float_sign_flip
5123 // value to get mask for all special values.
5124 movdqu(xtmp1, float_sign_flip, rscratch);
5125 vpcmpeqd(xtmp2, dst, xtmp1, Assembler::AVX_128bit);
5126 ptest(xtmp2, xtmp2);
5127 jccb(Assembler::equal, done);
5128
5129 // Flip float_sign_flip to get max integer value.
5130 vpcmpeqd(xtmp4, xtmp4, xtmp4, Assembler::AVX_128bit);
5131 pxor(xtmp1, xtmp4);
5132
5133 // Set detination lanes corresponding to unordered source lanes as zero.
5134 vpxor(xtmp4, xtmp4, xtmp4, src_vec_enc);
5135 vcmppd(xtmp3, src, src, Assembler::UNORD_Q, src_vec_enc);
5136
5137 // Shuffle mask vector and pack lower doubles word from each quadword lane.
5138 vector_crosslane_doubleword_pack_avx(xtmp3, xtmp3, xtmp4, xtmp5, 0x88, src_vec_enc);
5139 vblendvps(dst, dst, xtmp4, xtmp3, Assembler::AVX_128bit);
5140
5141 // Recompute the mask for remaining special value.
5142 pxor(xtmp2, xtmp3);
5143 // Extract mask corresponding to non-negative source lanes.
5144 vcmppd(xtmp3, src, xtmp4, Assembler::NLT_UQ, src_vec_enc);
5145
5146 // Shuffle mask vector and pack lower doubles word from each quadword lane.
5147 vector_crosslane_doubleword_pack_avx(xtmp3, xtmp3, xtmp4, xtmp5, 0x88, src_vec_enc);
5148 pand(xtmp3, xtmp2);
5149
5150 // Replace destination lanes holding special value(0x80000000) with max int
5151 // if corresponding source lane holds a +ve value.
5152 vblendvps(dst, dst, xtmp1, xtmp3, Assembler::AVX_128bit);
5153 bind(done);
5154 }
5155
5156
5157 void C2_MacroAssembler::vector_cast_int_to_subword(BasicType to_elem_bt, XMMRegister dst, XMMRegister zero,
5158 XMMRegister xtmp, Register rscratch, int vec_enc) {
5159 switch(to_elem_bt) {
5160 case T_SHORT:
5161 assert(rscratch != noreg || always_reachable(ExternalAddress(StubRoutines::x86::vector_int_to_short_mask())), "missing");
5162 vpand(dst, dst, ExternalAddress(StubRoutines::x86::vector_int_to_short_mask()), vec_enc, rscratch);
5163 vpackusdw(dst, dst, zero, vec_enc);
5164 if (vec_enc == Assembler::AVX_256bit) {
5165 vector_crosslane_doubleword_pack_avx(dst, dst, zero, xtmp, 0x44, vec_enc);
5166 }
5167 break;
5168 case T_BYTE:
5169 assert(rscratch != noreg || always_reachable(ExternalAddress(StubRoutines::x86::vector_int_to_byte_mask())), "missing");
5170 vpand(dst, dst, ExternalAddress(StubRoutines::x86::vector_int_to_byte_mask()), vec_enc, rscratch);
5171 vpackusdw(dst, dst, zero, vec_enc);
5172 if (vec_enc == Assembler::AVX_256bit) {
5173 vector_crosslane_doubleword_pack_avx(dst, dst, zero, xtmp, 0x44, vec_enc);
5174 }
5175 vpackuswb(dst, dst, zero, vec_enc);
5176 break;
5177 default: assert(false, "Unexpected basic type for target of vector cast int to subword: %s", type2name(to_elem_bt));
5178 }
5179 }
5180
5181 /*
5182 * Algorithm for vector D2L and F2I conversions (AVX 10.2 unsupported):-
5183 * a) Perform vector D2L/F2I cast.
5184 * b) Choose fast path if none of the result vector lane contains 0x80000000 value.
5185 * It signifies that source value could be any of the special floating point
5186 * values(NaN,-Inf,Inf,Max,-Min).
5187 * c) Set destination to zero if source is NaN value.
5188 * d) Replace 0x80000000 with MaxInt if source lane contains a +ve value.
5189 */
5190
5191 void C2_MacroAssembler::vector_castF2X_avx(BasicType to_elem_bt, XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5192 XMMRegister xtmp2, XMMRegister xtmp3, XMMRegister xtmp4,
5193 AddressLiteral float_sign_flip, Register rscratch, int vec_enc) {
5194 int to_elem_sz = type2aelembytes(to_elem_bt);
5195 assert(to_elem_sz <= 4, "");
5196 vcvttps2dq(dst, src, vec_enc);
5197 vector_cast_float_to_int_special_cases_avx(dst, src, xtmp1, xtmp2, xtmp3, xtmp4, rscratch, float_sign_flip, vec_enc);
5198 if (to_elem_sz < 4) {
5199 vpxor(xtmp4, xtmp4, xtmp4, vec_enc);
5200 vector_cast_int_to_subword(to_elem_bt, dst, xtmp4, xtmp3, rscratch, vec_enc);
5201 }
5202 }
5203
5204 void C2_MacroAssembler::vector_castF2X_evex(BasicType to_elem_bt, XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5205 XMMRegister xtmp2, KRegister ktmp1, KRegister ktmp2, AddressLiteral float_sign_flip,
5206 Register rscratch, int vec_enc) {
5207 int to_elem_sz = type2aelembytes(to_elem_bt);
5208 assert(to_elem_sz <= 4, "");
5209 vcvttps2dq(dst, src, vec_enc);
5210 vector_cast_float_to_int_special_cases_evex(dst, src, xtmp1, xtmp2, ktmp1, ktmp2, rscratch, float_sign_flip, vec_enc);
5211 switch(to_elem_bt) {
5212 case T_INT:
5213 break;
5214 case T_SHORT:
5215 evpmovdw(dst, dst, vec_enc);
5216 break;
5217 case T_BYTE:
5218 evpmovdb(dst, dst, vec_enc);
5219 break;
5220 default: assert(false, "Unexpected basic type for target of vector castF2X EVEX: %s", type2name(to_elem_bt));
5221 }
5222 }
5223
5224 void C2_MacroAssembler::vector_castF2L_evex(XMMRegister dst, XMMRegister src, XMMRegister xtmp1, XMMRegister xtmp2,
5225 KRegister ktmp1, KRegister ktmp2, AddressLiteral double_sign_flip,
5226 Register rscratch, int vec_enc) {
5227 evcvttps2qq(dst, src, vec_enc);
5228 vector_cast_float_to_long_special_cases_evex(dst, src, xtmp1, xtmp2, ktmp1, ktmp2, rscratch, double_sign_flip, vec_enc);
5229 }
5230
5231 // Handling for downcasting from double to integer or sub-word types on AVX2.
5232 void C2_MacroAssembler::vector_castD2X_avx(BasicType to_elem_bt, XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5233 XMMRegister xtmp2, XMMRegister xtmp3, XMMRegister xtmp4, XMMRegister xtmp5,
5234 AddressLiteral float_sign_flip, Register rscratch, int vec_enc) {
5235 int to_elem_sz = type2aelembytes(to_elem_bt);
5236 assert(to_elem_sz < 8, "");
5237 vcvttpd2dq(dst, src, vec_enc);
5238 vector_cast_double_to_int_special_cases_avx(dst, src, xtmp1, xtmp2, xtmp3, xtmp4, xtmp5, rscratch,
5239 float_sign_flip, vec_enc);
5240 if (to_elem_sz < 4) {
5241 // xtmp4 holds all zero lanes.
5242 vector_cast_int_to_subword(to_elem_bt, dst, xtmp4, xtmp5, rscratch, Assembler::AVX_128bit);
5243 }
5244 }
5245
5246 void C2_MacroAssembler::vector_castD2X_evex(BasicType to_elem_bt, XMMRegister dst, XMMRegister src,
5247 XMMRegister xtmp1, XMMRegister xtmp2, KRegister ktmp1,
5248 KRegister ktmp2, AddressLiteral sign_flip,
5249 Register rscratch, int vec_enc) {
5250 if (VM_Version::supports_avx512dq()) {
5251 evcvttpd2qq(dst, src, vec_enc);
5252 vector_cast_double_to_long_special_cases_evex(dst, src, xtmp1, xtmp2, ktmp1, ktmp2, rscratch, sign_flip, vec_enc);
5253 switch(to_elem_bt) {
5254 case T_LONG:
5255 break;
5256 case T_INT:
5257 evpmovsqd(dst, dst, vec_enc);
5258 break;
5259 case T_SHORT:
5260 evpmovsqd(dst, dst, vec_enc);
5261 evpmovdw(dst, dst, vec_enc);
5262 break;
5263 case T_BYTE:
5264 evpmovsqd(dst, dst, vec_enc);
5265 evpmovdb(dst, dst, vec_enc);
5266 break;
5267 default: assert(false, "Unexpected basic type for target of vector castD2X AVX512DQ EVEX: %s", type2name(to_elem_bt));
5268 }
5269 } else {
5270 assert(type2aelembytes(to_elem_bt) <= 4, "");
5271 vcvttpd2dq(dst, src, vec_enc);
5272 vector_cast_double_to_int_special_cases_evex(dst, src, xtmp1, xtmp2, ktmp1, ktmp2, rscratch, sign_flip, vec_enc);
5273 switch(to_elem_bt) {
5274 case T_INT:
5275 break;
5276 case T_SHORT:
5277 evpmovdw(dst, dst, vec_enc);
5278 break;
5279 case T_BYTE:
5280 evpmovdb(dst, dst, vec_enc);
5281 break;
5282 default: assert(false, "Unexpected basic type for target of vector castD2X EVEX: %s", type2name(to_elem_bt));
5283 }
5284 }
5285 }
5286
5287 void C2_MacroAssembler::vector_castF2X_avx10_2(BasicType to_elem_bt, XMMRegister dst, XMMRegister src, int vec_enc) {
5288 switch(to_elem_bt) {
5289 case T_LONG:
5290 evcvttps2qqs(dst, src, vec_enc);
5291 break;
5292 case T_INT:
5293 evcvttps2dqs(dst, src, vec_enc);
5294 break;
5295 case T_SHORT:
5296 evcvttps2dqs(dst, src, vec_enc);
5297 evpmovdw(dst, dst, vec_enc);
5298 break;
5299 case T_BYTE:
5300 evcvttps2dqs(dst, src, vec_enc);
5301 evpmovdb(dst, dst, vec_enc);
5302 break;
5303 default: assert(false, "Unexpected basic type for target of vector castF2X AVX10 (reg src): %s", type2name(to_elem_bt));
5304 }
5305 }
5306
5307 void C2_MacroAssembler::vector_castF2X_avx10_2(BasicType to_elem_bt, XMMRegister dst, Address src, int vec_enc) {
5308 switch(to_elem_bt) {
5309 case T_LONG:
5310 evcvttps2qqs(dst, src, vec_enc);
5311 break;
5312 case T_INT:
5313 evcvttps2dqs(dst, src, vec_enc);
5314 break;
5315 case T_SHORT:
5316 evcvttps2dqs(dst, src, vec_enc);
5317 evpmovdw(dst, dst, vec_enc);
5318 break;
5319 case T_BYTE:
5320 evcvttps2dqs(dst, src, vec_enc);
5321 evpmovdb(dst, dst, vec_enc);
5322 break;
5323 default: assert(false, "Unexpected basic type for target of vector castF2X AVX10 (mem src): %s", type2name(to_elem_bt));
5324 }
5325 }
5326
5327 void C2_MacroAssembler::vector_castD2X_avx10_2(BasicType to_elem_bt, XMMRegister dst, XMMRegister src, int vec_enc) {
5328 switch(to_elem_bt) {
5329 case T_LONG:
5330 evcvttpd2qqs(dst, src, vec_enc);
5331 break;
5332 case T_INT:
5333 evcvttpd2dqs(dst, src, vec_enc);
5334 break;
5335 case T_SHORT:
5336 evcvttpd2dqs(dst, src, vec_enc);
5337 evpmovdw(dst, dst, vec_enc);
5338 break;
5339 case T_BYTE:
5340 evcvttpd2dqs(dst, src, vec_enc);
5341 evpmovdb(dst, dst, vec_enc);
5342 break;
5343 default: assert(false, "Unexpected basic type for target of vector castD2X AVX10 (reg src): %s", type2name(to_elem_bt));
5344 }
5345 }
5346
5347 void C2_MacroAssembler::vector_castD2X_avx10_2(BasicType to_elem_bt, XMMRegister dst, Address src, int vec_enc) {
5348 switch(to_elem_bt) {
5349 case T_LONG:
5350 evcvttpd2qqs(dst, src, vec_enc);
5351 break;
5352 case T_INT:
5353 evcvttpd2dqs(dst, src, vec_enc);
5354 break;
5355 case T_SHORT:
5356 evcvttpd2dqs(dst, src, vec_enc);
5357 evpmovdw(dst, dst, vec_enc);
5358 break;
5359 case T_BYTE:
5360 evcvttpd2dqs(dst, src, vec_enc);
5361 evpmovdb(dst, dst, vec_enc);
5362 break;
5363 default: assert(false, "Unexpected basic type for target of vector castD2X AVX10 (mem src): %s", type2name(to_elem_bt));
5364 }
5365 }
5366
5367 void C2_MacroAssembler::vector_round_double_evex(XMMRegister dst, XMMRegister src,
5368 AddressLiteral double_sign_flip, AddressLiteral new_mxcsr, int vec_enc,
5369 Register tmp, XMMRegister xtmp1, XMMRegister xtmp2, KRegister ktmp1, KRegister ktmp2) {
5370 // Perform floor(val+0.5) operation under the influence of MXCSR.RC mode roundTowards -inf.
5371 // and re-instantiate original MXCSR.RC mode after that.
5372 ldmxcsr(new_mxcsr, tmp /*rscratch*/);
5373
5374 mov64(tmp, julong_cast(0.5L));
5375 evpbroadcastq(xtmp1, tmp, vec_enc);
5376 vaddpd(xtmp1, src , xtmp1, vec_enc);
5377 evcvtpd2qq(dst, xtmp1, vec_enc);
5378 vector_cast_double_to_long_special_cases_evex(dst, src, xtmp1, xtmp2, ktmp1, ktmp2, tmp /*rscratch*/,
5379 double_sign_flip, vec_enc);;
5380
5381 ldmxcsr(ExternalAddress(StubRoutines::x86::addr_mxcsr_std()), tmp /*rscratch*/);
5382 }
5383
5384 void C2_MacroAssembler::vector_round_float_evex(XMMRegister dst, XMMRegister src,
5385 AddressLiteral float_sign_flip, AddressLiteral new_mxcsr, int vec_enc,
5386 Register tmp, XMMRegister xtmp1, XMMRegister xtmp2, KRegister ktmp1, KRegister ktmp2) {
5387 // Perform floor(val+0.5) operation under the influence of MXCSR.RC mode roundTowards -inf.
5388 // and re-instantiate original MXCSR.RC mode after that.
5389 ldmxcsr(new_mxcsr, tmp /*rscratch*/);
5390
5391 movl(tmp, jint_cast(0.5));
5392 movq(xtmp1, tmp);
5393 vbroadcastss(xtmp1, xtmp1, vec_enc);
5394 vaddps(xtmp1, src , xtmp1, vec_enc);
5395 vcvtps2dq(dst, xtmp1, vec_enc);
5396 vector_cast_float_to_int_special_cases_evex(dst, src, xtmp1, xtmp2, ktmp1, ktmp2, tmp /*rscratch*/,
5397 float_sign_flip, vec_enc);
5398
5399 ldmxcsr(ExternalAddress(StubRoutines::x86::addr_mxcsr_std()), tmp /*rscratch*/);
5400 }
5401
5402 void C2_MacroAssembler::vector_round_float_avx(XMMRegister dst, XMMRegister src,
5403 AddressLiteral float_sign_flip, AddressLiteral new_mxcsr, int vec_enc,
5404 Register tmp, XMMRegister xtmp1, XMMRegister xtmp2, XMMRegister xtmp3, XMMRegister xtmp4) {
5405 // Perform floor(val+0.5) operation under the influence of MXCSR.RC mode roundTowards -inf.
5406 // and re-instantiate original MXCSR.RC mode after that.
5407 ldmxcsr(new_mxcsr, tmp /*rscratch*/);
5408
5409 movl(tmp, jint_cast(0.5));
5410 movq(xtmp1, tmp);
5411 vbroadcastss(xtmp1, xtmp1, vec_enc);
5412 vaddps(xtmp1, src , xtmp1, vec_enc);
5413 vcvtps2dq(dst, xtmp1, vec_enc);
5414 vector_cast_float_to_int_special_cases_avx(dst, src, xtmp1, xtmp2, xtmp3, xtmp4, tmp /*rscratch*/, float_sign_flip, vec_enc);
5415
5416 ldmxcsr(ExternalAddress(StubRoutines::x86::addr_mxcsr_std()), tmp /*rscratch*/);
5417 }
5418
5419 void C2_MacroAssembler::vector_unsigned_cast(XMMRegister dst, XMMRegister src, int vlen_enc,
5420 BasicType from_elem_bt, BasicType to_elem_bt) {
5421 switch (from_elem_bt) {
5422 case T_BYTE:
5423 switch (to_elem_bt) {
5424 case T_SHORT: vpmovzxbw(dst, src, vlen_enc); break;
5425 case T_INT: vpmovzxbd(dst, src, vlen_enc); break;
5426 case T_LONG: vpmovzxbq(dst, src, vlen_enc); break;
5427 default: ShouldNotReachHere();
5428 }
5429 break;
5430 case T_SHORT:
5431 switch (to_elem_bt) {
5432 case T_INT: vpmovzxwd(dst, src, vlen_enc); break;
5433 case T_LONG: vpmovzxwq(dst, src, vlen_enc); break;
5434 default: ShouldNotReachHere();
5435 }
5436 break;
5437 case T_INT:
5438 assert(to_elem_bt == T_LONG, "");
5439 vpmovzxdq(dst, src, vlen_enc);
5440 break;
5441 default:
5442 ShouldNotReachHere();
5443 }
5444 }
5445
5446 void C2_MacroAssembler::vector_signed_cast(XMMRegister dst, XMMRegister src, int vlen_enc,
5447 BasicType from_elem_bt, BasicType to_elem_bt) {
5448 switch (from_elem_bt) {
5449 case T_BYTE:
5450 switch (to_elem_bt) {
5451 case T_SHORT: vpmovsxbw(dst, src, vlen_enc); break;
5452 case T_INT: vpmovsxbd(dst, src, vlen_enc); break;
5453 case T_LONG: vpmovsxbq(dst, src, vlen_enc); break;
5454 default: ShouldNotReachHere();
5455 }
5456 break;
5457 case T_SHORT:
5458 switch (to_elem_bt) {
5459 case T_INT: vpmovsxwd(dst, src, vlen_enc); break;
5460 case T_LONG: vpmovsxwq(dst, src, vlen_enc); break;
5461 default: ShouldNotReachHere();
5462 }
5463 break;
5464 case T_INT:
5465 assert(to_elem_bt == T_LONG, "");
5466 vpmovsxdq(dst, src, vlen_enc);
5467 break;
5468 default:
5469 ShouldNotReachHere();
5470 }
5471 }
5472
5473 void C2_MacroAssembler::vector_mask_cast(XMMRegister dst, XMMRegister src,
5474 BasicType dst_bt, BasicType src_bt, int vlen) {
5475 int vlen_enc = vector_length_encoding(MAX2(type2aelembytes(src_bt), type2aelembytes(dst_bt)) * vlen);
5476 assert(vlen_enc != AVX_512bit, "");
5477
5478 int dst_bt_size = type2aelembytes(dst_bt);
5479 int src_bt_size = type2aelembytes(src_bt);
5480 if (dst_bt_size > src_bt_size) {
5481 switch (dst_bt_size / src_bt_size) {
5482 case 2: vpmovsxbw(dst, src, vlen_enc); break;
5483 case 4: vpmovsxbd(dst, src, vlen_enc); break;
5484 case 8: vpmovsxbq(dst, src, vlen_enc); break;
5485 default: ShouldNotReachHere();
5486 }
5487 } else {
5488 assert(dst_bt_size < src_bt_size, "");
5489 switch (src_bt_size / dst_bt_size) {
5490 case 2: {
5491 if (vlen_enc == AVX_128bit) {
5492 vpacksswb(dst, src, src, vlen_enc);
5493 } else {
5494 vpacksswb(dst, src, src, vlen_enc);
5495 vpermq(dst, dst, 0x08, vlen_enc);
5496 }
5497 break;
5498 }
5499 case 4: {
5500 if (vlen_enc == AVX_128bit) {
5501 vpackssdw(dst, src, src, vlen_enc);
5502 vpacksswb(dst, dst, dst, vlen_enc);
5503 } else {
5504 vpackssdw(dst, src, src, vlen_enc);
5505 vpermq(dst, dst, 0x08, vlen_enc);
5506 vpacksswb(dst, dst, dst, AVX_128bit);
5507 }
5508 break;
5509 }
5510 case 8: {
5511 if (vlen_enc == AVX_128bit) {
5512 vpshufd(dst, src, 0x08, vlen_enc);
5513 vpackssdw(dst, dst, dst, vlen_enc);
5514 vpacksswb(dst, dst, dst, vlen_enc);
5515 } else {
5516 vpshufd(dst, src, 0x08, vlen_enc);
5517 vpermq(dst, dst, 0x08, vlen_enc);
5518 vpackssdw(dst, dst, dst, AVX_128bit);
5519 vpacksswb(dst, dst, dst, AVX_128bit);
5520 }
5521 break;
5522 }
5523 default: ShouldNotReachHere();
5524 }
5525 }
5526 }
5527
5528 void C2_MacroAssembler::evpternlog(XMMRegister dst, int func, KRegister mask, XMMRegister src2, XMMRegister src3,
5529 bool merge, BasicType bt, int vlen_enc) {
5530 if (bt == T_INT) {
5531 evpternlogd(dst, func, mask, src2, src3, merge, vlen_enc);
5532 } else {
5533 assert(bt == T_LONG, "");
5534 evpternlogq(dst, func, mask, src2, src3, merge, vlen_enc);
5535 }
5536 }
5537
5538 void C2_MacroAssembler::evpternlog(XMMRegister dst, int func, KRegister mask, XMMRegister src2, Address src3,
5539 bool merge, BasicType bt, int vlen_enc) {
5540 if (bt == T_INT) {
5541 evpternlogd(dst, func, mask, src2, src3, merge, vlen_enc);
5542 } else {
5543 assert(bt == T_LONG, "");
5544 evpternlogq(dst, func, mask, src2, src3, merge, vlen_enc);
5545 }
5546 }
5547
5548 void C2_MacroAssembler::vector_long_to_maskvec(XMMRegister dst, Register src, Register rtmp1,
5549 Register rtmp2, XMMRegister xtmp, int mask_len,
5550 int vec_enc) {
5551 int index = 0;
5552 int vindex = 0;
5553 mov64(rtmp1, 0x0101010101010101L);
5554 pdepq(rtmp1, src, rtmp1);
5555 if (mask_len > 8) {
5556 movq(rtmp2, src);
5557 vpxor(xtmp, xtmp, xtmp, vec_enc);
5558 movq(xtmp, rtmp1);
5559 }
5560 movq(dst, rtmp1);
5561
5562 mask_len -= 8;
5563 while (mask_len > 0) {
5564 assert ((mask_len & 0x7) == 0, "mask must be multiple of 8");
5565 index++;
5566 if ((index % 2) == 0) {
5567 pxor(xtmp, xtmp);
5568 }
5569 mov64(rtmp1, 0x0101010101010101L);
5570 shrq(rtmp2, 8);
5571 pdepq(rtmp1, rtmp2, rtmp1);
5572 pinsrq(xtmp, rtmp1, index % 2);
5573 vindex = index / 2;
5574 if (vindex) {
5575 // Write entire 16 byte vector when both 64 bit
5576 // lanes are update to save redundant instructions.
5577 if (index % 2) {
5578 vinsertf128(dst, dst, xtmp, vindex);
5579 }
5580 } else {
5581 vmovdqu(dst, xtmp);
5582 }
5583 mask_len -= 8;
5584 }
5585 }
5586
5587 void C2_MacroAssembler::vector_mask_operation_helper(int opc, Register dst, Register tmp, int masklen) {
5588 switch(opc) {
5589 case Op_VectorMaskTrueCount:
5590 popcntq(dst, tmp);
5591 break;
5592 case Op_VectorMaskLastTrue:
5593 if (VM_Version::supports_lzcnt()) {
5594 lzcntq(tmp, tmp);
5595 movl(dst, 63);
5596 subl(dst, tmp);
5597 } else {
5598 movl(dst, -1);
5599 bsrq(tmp, tmp);
5600 cmov32(Assembler::notZero, dst, tmp);
5601 }
5602 break;
5603 case Op_VectorMaskFirstTrue:
5604 if (UseCountTrailingZerosInstruction) {
5605 if (masklen < 32) {
5606 orl(tmp, 1 << masklen);
5607 tzcntl(dst, tmp);
5608 } else if (masklen == 32) {
5609 tzcntl(dst, tmp);
5610 } else {
5611 assert(masklen == 64, "");
5612 tzcntq(dst, tmp);
5613 }
5614 } else {
5615 if (masklen < 32) {
5616 orl(tmp, 1 << masklen);
5617 bsfl(dst, tmp);
5618 } else {
5619 assert(masklen == 32 || masklen == 64, "");
5620 movl(dst, masklen);
5621 if (masklen == 32) {
5622 bsfl(tmp, tmp);
5623 } else {
5624 bsfq(tmp, tmp);
5625 }
5626 cmov32(Assembler::notZero, dst, tmp);
5627 }
5628 }
5629 break;
5630 case Op_VectorMaskToLong:
5631 assert(dst == tmp, "Dst and tmp should be the same for toLong operations");
5632 break;
5633 default: assert(false, "Unhandled mask operation");
5634 }
5635 }
5636
5637 void C2_MacroAssembler::vector_mask_operation(int opc, Register dst, KRegister mask, Register tmp,
5638 int masklen, int masksize, int vec_enc) {
5639 assert(VM_Version::supports_popcnt(), "");
5640
5641 if(VM_Version::supports_avx512bw()) {
5642 kmovql(tmp, mask);
5643 } else {
5644 assert(masklen <= 16, "");
5645 kmovwl(tmp, mask);
5646 }
5647
5648 // Mask generated out of partial vector comparisons/replicate/mask manipulation
5649 // operations needs to be clipped.
5650 if (masksize < 16 && opc != Op_VectorMaskFirstTrue) {
5651 andq(tmp, (1 << masklen) - 1);
5652 }
5653
5654 vector_mask_operation_helper(opc, dst, tmp, masklen);
5655 }
5656
5657 void C2_MacroAssembler::vector_mask_operation(int opc, Register dst, XMMRegister mask, XMMRegister xtmp,
5658 Register tmp, int masklen, BasicType bt, int vec_enc) {
5659 assert((vec_enc == AVX_128bit && VM_Version::supports_avx()) ||
5660 (vec_enc == AVX_256bit && (VM_Version::supports_avx2() || type2aelembytes(bt) >= 4)), "");
5661 assert(VM_Version::supports_popcnt(), "");
5662
5663 bool need_clip = false;
5664 switch(bt) {
5665 case T_BOOLEAN:
5666 // While masks of other types contain 0, -1; boolean masks contain lane values of 0, 1
5667 vpxor(xtmp, xtmp, xtmp, vec_enc);
5668 vpsubb(xtmp, xtmp, mask, vec_enc);
5669 vpmovmskb(tmp, xtmp, vec_enc);
5670 need_clip = masklen < 16;
5671 break;
5672 case T_BYTE:
5673 vpmovmskb(tmp, mask, vec_enc);
5674 need_clip = masklen < 16;
5675 break;
5676 case T_SHORT:
5677 vpacksswb(xtmp, mask, mask, vec_enc);
5678 if (masklen >= 16) {
5679 vpermpd(xtmp, xtmp, 8, vec_enc);
5680 }
5681 vpmovmskb(tmp, xtmp, Assembler::AVX_128bit);
5682 need_clip = masklen < 16;
5683 break;
5684 case T_INT:
5685 case T_FLOAT:
5686 vmovmskps(tmp, mask, vec_enc);
5687 need_clip = masklen < 4;
5688 break;
5689 case T_LONG:
5690 case T_DOUBLE:
5691 vmovmskpd(tmp, mask, vec_enc);
5692 need_clip = masklen < 2;
5693 break;
5694 default: assert(false, "Unhandled type, %s", type2name(bt));
5695 }
5696
5697 // Mask generated out of partial vector comparisons/replicate/mask manipulation
5698 // operations needs to be clipped.
5699 if (need_clip && opc != Op_VectorMaskFirstTrue) {
5700 // need_clip implies masklen < 32
5701 andq(tmp, (1 << masklen) - 1);
5702 }
5703
5704 vector_mask_operation_helper(opc, dst, tmp, masklen);
5705 }
5706
5707 void C2_MacroAssembler::vector_mask_compress(KRegister dst, KRegister src, Register rtmp1,
5708 Register rtmp2, int mask_len) {
5709 kmov(rtmp1, src);
5710 andq(rtmp1, (0xFFFFFFFFFFFFFFFFUL >> (64 - mask_len)));
5711 mov64(rtmp2, -1L);
5712 pextq(rtmp2, rtmp2, rtmp1);
5713 kmov(dst, rtmp2);
5714 }
5715
5716 void C2_MacroAssembler::vector_compress_expand_avx2(int opcode, XMMRegister dst, XMMRegister src,
5717 XMMRegister mask, Register rtmp, Register rscratch,
5718 XMMRegister permv, XMMRegister xtmp, BasicType bt,
5719 int vec_enc) {
5720 assert(type2aelembytes(bt) >= 4, "");
5721 assert(opcode == Op_CompressV || opcode == Op_ExpandV, "");
5722 address compress_perm_table = nullptr;
5723 address expand_perm_table = nullptr;
5724 if (type2aelembytes(bt) == 8) {
5725 compress_perm_table = StubRoutines::x86::compress_perm_table64();
5726 expand_perm_table = StubRoutines::x86::expand_perm_table64();
5727 vmovmskpd(rtmp, mask, vec_enc);
5728 } else {
5729 compress_perm_table = StubRoutines::x86::compress_perm_table32();
5730 expand_perm_table = StubRoutines::x86::expand_perm_table32();
5731 vmovmskps(rtmp, mask, vec_enc);
5732 }
5733 shlq(rtmp, 5); // for 32 byte permute row.
5734 if (opcode == Op_CompressV) {
5735 lea(rscratch, ExternalAddress(compress_perm_table));
5736 } else {
5737 lea(rscratch, ExternalAddress(expand_perm_table));
5738 }
5739 addptr(rtmp, rscratch);
5740 vmovdqu(permv, Address(rtmp));
5741 vpermps(dst, permv, src, Assembler::AVX_256bit);
5742 vpxor(xtmp, xtmp, xtmp, vec_enc);
5743 // Blend the result with zero vector using permute mask, each column entry
5744 // in a permute table row contains either a valid permute index or a -1 (default)
5745 // value, this can potentially be used as a blending mask after
5746 // compressing/expanding the source vector lanes.
5747 vblendvps(dst, dst, xtmp, permv, vec_enc, true, permv);
5748 }
5749
5750 void C2_MacroAssembler::vector_compress_expand(int opcode, XMMRegister dst, XMMRegister src, KRegister mask,
5751 bool merge, BasicType bt, int vec_enc) {
5752 if (opcode == Op_CompressV) {
5753 switch(bt) {
5754 case T_BYTE:
5755 evpcompressb(dst, mask, src, merge, vec_enc);
5756 break;
5757 case T_CHAR:
5758 case T_SHORT:
5759 evpcompressw(dst, mask, src, merge, vec_enc);
5760 break;
5761 case T_INT:
5762 evpcompressd(dst, mask, src, merge, vec_enc);
5763 break;
5764 case T_FLOAT:
5765 evcompressps(dst, mask, src, merge, vec_enc);
5766 break;
5767 case T_LONG:
5768 evpcompressq(dst, mask, src, merge, vec_enc);
5769 break;
5770 case T_DOUBLE:
5771 evcompresspd(dst, mask, src, merge, vec_enc);
5772 break;
5773 default:
5774 fatal("Unsupported type %s", type2name(bt));
5775 break;
5776 }
5777 } else {
5778 assert(opcode == Op_ExpandV, "");
5779 switch(bt) {
5780 case T_BYTE:
5781 evpexpandb(dst, mask, src, merge, vec_enc);
5782 break;
5783 case T_CHAR:
5784 case T_SHORT:
5785 evpexpandw(dst, mask, src, merge, vec_enc);
5786 break;
5787 case T_INT:
5788 evpexpandd(dst, mask, src, merge, vec_enc);
5789 break;
5790 case T_FLOAT:
5791 evexpandps(dst, mask, src, merge, vec_enc);
5792 break;
5793 case T_LONG:
5794 evpexpandq(dst, mask, src, merge, vec_enc);
5795 break;
5796 case T_DOUBLE:
5797 evexpandpd(dst, mask, src, merge, vec_enc);
5798 break;
5799 default:
5800 fatal("Unsupported type %s", type2name(bt));
5801 break;
5802 }
5803 }
5804 }
5805
5806 void C2_MacroAssembler::vector_signum_evex(int opcode, XMMRegister dst, XMMRegister src, XMMRegister zero, XMMRegister one,
5807 KRegister ktmp1, int vec_enc) {
5808 if (opcode == Op_SignumVD) {
5809 vsubpd(dst, zero, one, vec_enc);
5810 // if src < 0 ? -1 : 1
5811 evcmppd(ktmp1, k0, src, zero, Assembler::LT_OQ, vec_enc);
5812 evblendmpd(dst, ktmp1, one, dst, true, vec_enc);
5813 // if src == NaN, -0.0 or 0.0 return src.
5814 evcmppd(ktmp1, k0, src, zero, Assembler::EQ_UQ, vec_enc);
5815 evblendmpd(dst, ktmp1, dst, src, true, vec_enc);
5816 } else {
5817 assert(opcode == Op_SignumVF, "");
5818 vsubps(dst, zero, one, vec_enc);
5819 // if src < 0 ? -1 : 1
5820 evcmpps(ktmp1, k0, src, zero, Assembler::LT_OQ, vec_enc);
5821 evblendmps(dst, ktmp1, one, dst, true, vec_enc);
5822 // if src == NaN, -0.0 or 0.0 return src.
5823 evcmpps(ktmp1, k0, src, zero, Assembler::EQ_UQ, vec_enc);
5824 evblendmps(dst, ktmp1, dst, src, true, vec_enc);
5825 }
5826 }
5827
5828 void C2_MacroAssembler::vector_signum_avx(int opcode, XMMRegister dst, XMMRegister src, XMMRegister zero, XMMRegister one,
5829 XMMRegister xtmp1, int vec_enc) {
5830 if (opcode == Op_SignumVD) {
5831 vsubpd(dst, zero, one, vec_enc);
5832 // if src < 0 ? -1 : 1
5833 vblendvpd(dst, one, dst, src, vec_enc, true, xtmp1);
5834 // if src == NaN, -0.0 or 0.0 return src.
5835 vcmppd(xtmp1, src, zero, Assembler::EQ_UQ, vec_enc);
5836 vblendvpd(dst, dst, src, xtmp1, vec_enc, false, xtmp1);
5837 } else {
5838 assert(opcode == Op_SignumVF, "");
5839 vsubps(dst, zero, one, vec_enc);
5840 // if src < 0 ? -1 : 1
5841 vblendvps(dst, one, dst, src, vec_enc, true, xtmp1);
5842 // if src == NaN, -0.0 or 0.0 return src.
5843 vcmpps(xtmp1, src, zero, Assembler::EQ_UQ, vec_enc);
5844 vblendvps(dst, dst, src, xtmp1, vec_enc, false, xtmp1);
5845 }
5846 }
5847
5848 void C2_MacroAssembler::vector_maskall_operation(KRegister dst, Register src, int mask_len) {
5849 if (VM_Version::supports_avx512bw()) {
5850 if (mask_len > 32) {
5851 kmovql(dst, src);
5852 } else {
5853 kmovdl(dst, src);
5854 if (mask_len != 32) {
5855 kshiftrdl(dst, dst, 32 - mask_len);
5856 }
5857 }
5858 } else {
5859 assert(mask_len <= 16, "");
5860 kmovwl(dst, src);
5861 if (mask_len != 16) {
5862 kshiftrwl(dst, dst, 16 - mask_len);
5863 }
5864 }
5865 }
5866
5867 void C2_MacroAssembler::vbroadcast(BasicType bt, XMMRegister dst, int imm32, Register rtmp, int vec_enc) {
5868 int lane_size = type2aelembytes(bt);
5869 if ((is_non_subword_integral_type(bt) && VM_Version::supports_avx512vl()) ||
5870 (is_subword_type(bt) && VM_Version::supports_avx512vlbw())) {
5871 movptr(rtmp, imm32);
5872 switch(lane_size) {
5873 case 1 : evpbroadcastb(dst, rtmp, vec_enc); break;
5874 case 2 : evpbroadcastw(dst, rtmp, vec_enc); break;
5875 case 4 : evpbroadcastd(dst, rtmp, vec_enc); break;
5876 case 8 : evpbroadcastq(dst, rtmp, vec_enc); break;
5877 fatal("Unsupported lane size %d", lane_size);
5878 break;
5879 }
5880 } else {
5881 movptr(rtmp, imm32);
5882 movq(dst, rtmp);
5883 switch(lane_size) {
5884 case 1 : vpbroadcastb(dst, dst, vec_enc); break;
5885 case 2 : vpbroadcastw(dst, dst, vec_enc); break;
5886 case 4 : vpbroadcastd(dst, dst, vec_enc); break;
5887 case 8 : vpbroadcastq(dst, dst, vec_enc); break;
5888 fatal("Unsupported lane size %d", lane_size);
5889 break;
5890 }
5891 }
5892 }
5893
5894 //
5895 // Following is lookup table based popcount computation algorithm:-
5896 // Index Bit set count
5897 // [ 0000 -> 0,
5898 // 0001 -> 1,
5899 // 0010 -> 1,
5900 // 0011 -> 2,
5901 // 0100 -> 1,
5902 // 0101 -> 2,
5903 // 0110 -> 2,
5904 // 0111 -> 3,
5905 // 1000 -> 1,
5906 // 1001 -> 2,
5907 // 1010 -> 3,
5908 // 1011 -> 3,
5909 // 1100 -> 2,
5910 // 1101 -> 3,
5911 // 1111 -> 4 ]
5912 // a. Count the number of 1s in 4 LSB bits of each byte. These bits are used as
5913 // shuffle indices for lookup table access.
5914 // b. Right shift each byte of vector lane by 4 positions.
5915 // c. Count the number of 1s in 4 MSB bits each byte. These bits are used as
5916 // shuffle indices for lookup table access.
5917 // d. Add the bitset count of upper and lower 4 bits of each byte.
5918 // e. Unpack double words to quad words and compute sum of absolute difference of bitset
5919 // count of all the bytes of a quadword.
5920 // f. Perform step e. for upper 128bit vector lane.
5921 // g. Pack the bitset count of quadwords back to double word.
5922 // h. Unpacking and packing operations are not needed for 64bit vector lane.
5923
5924 void C2_MacroAssembler::vector_popcount_byte(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5925 XMMRegister xtmp2, Register rtmp, int vec_enc) {
5926 assert((vec_enc == Assembler::AVX_512bit && VM_Version::supports_avx512bw()) || VM_Version::supports_avx2(), "");
5927 vbroadcast(T_INT, xtmp1, 0x0F0F0F0F, rtmp, vec_enc);
5928 vpsrlw(dst, src, 4, vec_enc);
5929 vpand(dst, dst, xtmp1, vec_enc);
5930 vpand(xtmp1, src, xtmp1, vec_enc);
5931 vmovdqu(xtmp2, ExternalAddress(StubRoutines::x86::vector_popcount_lut()), vec_enc, noreg);
5932 vpshufb(xtmp1, xtmp2, xtmp1, vec_enc);
5933 vpshufb(dst, xtmp2, dst, vec_enc);
5934 vpaddb(dst, dst, xtmp1, vec_enc);
5935 }
5936
5937 void C2_MacroAssembler::vector_popcount_int(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5938 XMMRegister xtmp2, Register rtmp, int vec_enc) {
5939 vector_popcount_byte(xtmp1, src, dst, xtmp2, rtmp, vec_enc);
5940 // Following code is as per steps e,f,g and h of above algorithm.
5941 vpxor(xtmp2, xtmp2, xtmp2, vec_enc);
5942 vpunpckhdq(dst, xtmp1, xtmp2, vec_enc);
5943 vpsadbw(dst, dst, xtmp2, vec_enc);
5944 vpunpckldq(xtmp1, xtmp1, xtmp2, vec_enc);
5945 vpsadbw(xtmp1, xtmp1, xtmp2, vec_enc);
5946 vpackuswb(dst, xtmp1, dst, vec_enc);
5947 }
5948
5949 void C2_MacroAssembler::vector_popcount_short(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5950 XMMRegister xtmp2, Register rtmp, int vec_enc) {
5951 vector_popcount_byte(xtmp1, src, dst, xtmp2, rtmp, vec_enc);
5952 // Add the popcount of upper and lower bytes of word.
5953 vbroadcast(T_INT, xtmp2, 0x00FF00FF, rtmp, vec_enc);
5954 vpsrlw(dst, xtmp1, 8, vec_enc);
5955 vpand(xtmp1, xtmp1, xtmp2, vec_enc);
5956 vpaddw(dst, dst, xtmp1, vec_enc);
5957 }
5958
5959 void C2_MacroAssembler::vector_popcount_long(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5960 XMMRegister xtmp2, Register rtmp, int vec_enc) {
5961 vector_popcount_byte(xtmp1, src, dst, xtmp2, rtmp, vec_enc);
5962 vpxor(xtmp2, xtmp2, xtmp2, vec_enc);
5963 vpsadbw(dst, xtmp1, xtmp2, vec_enc);
5964 }
5965
5966 void C2_MacroAssembler::vector_popcount_integral(BasicType bt, XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
5967 XMMRegister xtmp2, Register rtmp, int vec_enc) {
5968 switch(bt) {
5969 case T_LONG:
5970 vector_popcount_long(dst, src, xtmp1, xtmp2, rtmp, vec_enc);
5971 break;
5972 case T_INT:
5973 vector_popcount_int(dst, src, xtmp1, xtmp2, rtmp, vec_enc);
5974 break;
5975 case T_CHAR:
5976 case T_SHORT:
5977 vector_popcount_short(dst, src, xtmp1, xtmp2, rtmp, vec_enc);
5978 break;
5979 case T_BYTE:
5980 case T_BOOLEAN:
5981 vector_popcount_byte(dst, src, xtmp1, xtmp2, rtmp, vec_enc);
5982 break;
5983 default:
5984 fatal("Unsupported type %s", type2name(bt));
5985 break;
5986 }
5987 }
5988
5989 void C2_MacroAssembler::vector_popcount_integral_evex(BasicType bt, XMMRegister dst, XMMRegister src,
5990 KRegister mask, bool merge, int vec_enc) {
5991 assert(VM_Version::supports_avx512vl() || vec_enc == Assembler::AVX_512bit, "");
5992 switch(bt) {
5993 case T_LONG:
5994 assert(VM_Version::supports_avx512_vpopcntdq(), "");
5995 evpopcntq(dst, mask, src, merge, vec_enc);
5996 break;
5997 case T_INT:
5998 assert(VM_Version::supports_avx512_vpopcntdq(), "");
5999 evpopcntd(dst, mask, src, merge, vec_enc);
6000 break;
6001 case T_CHAR:
6002 case T_SHORT:
6003 assert(VM_Version::supports_avx512_bitalg(), "");
6004 evpopcntw(dst, mask, src, merge, vec_enc);
6005 break;
6006 case T_BYTE:
6007 case T_BOOLEAN:
6008 assert(VM_Version::supports_avx512_bitalg(), "");
6009 evpopcntb(dst, mask, src, merge, vec_enc);
6010 break;
6011 default:
6012 fatal("Unsupported type %s", type2name(bt));
6013 break;
6014 }
6015 }
6016
6017 // Bit reversal algorithm first reverses the bits of each byte followed by
6018 // a byte level reversal for multi-byte primitive types (short/int/long).
6019 // Algorithm performs a lookup table access to get reverse bit sequence
6020 // corresponding to a 4 bit value. Thus a reverse bit sequence for a byte
6021 // is obtained by swapping the reverse bit sequences of upper and lower
6022 // nibble of a byte.
6023 void C2_MacroAssembler::vector_reverse_bit(BasicType bt, XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
6024 XMMRegister xtmp2, Register rtmp, int vec_enc) {
6025 if (VM_Version::supports_avx512vlbw()) {
6026
6027 // Get the reverse bit sequence of lower nibble of each byte.
6028 vmovdqu(xtmp1, ExternalAddress(StubRoutines::x86::vector_reverse_bit_lut()), vec_enc, noreg);
6029 vbroadcast(T_INT, xtmp2, 0x0F0F0F0F, rtmp, vec_enc);
6030 evpandq(dst, xtmp2, src, vec_enc);
6031 vpshufb(dst, xtmp1, dst, vec_enc);
6032 vpsllq(dst, dst, 4, vec_enc);
6033
6034 // Get the reverse bit sequence of upper nibble of each byte.
6035 vpandn(xtmp2, xtmp2, src, vec_enc);
6036 vpsrlq(xtmp2, xtmp2, 4, vec_enc);
6037 vpshufb(xtmp2, xtmp1, xtmp2, vec_enc);
6038
6039 // Perform logical OR operation b/w left shifted reverse bit sequence of lower nibble and
6040 // right shifted reverse bit sequence of upper nibble to obtain the reverse bit sequence of each byte.
6041 evporq(xtmp2, dst, xtmp2, vec_enc);
6042 vector_reverse_byte(bt, dst, xtmp2, vec_enc);
6043
6044 } else if(vec_enc == Assembler::AVX_512bit) {
6045 // Shift based bit reversal.
6046 assert(bt == T_LONG || bt == T_INT, "");
6047
6048 // Swap lower and upper nibble of each byte.
6049 vector_swap_nbits(4, 0x0F0F0F0F, xtmp1, src, xtmp2, rtmp, vec_enc);
6050
6051 // Swap two least and most significant bits of each nibble.
6052 vector_swap_nbits(2, 0x33333333, dst, xtmp1, xtmp2, rtmp, vec_enc);
6053
6054 // Swap adjacent pair of bits.
6055 evmovdqul(xtmp1, k0, dst, true, vec_enc);
6056 vector_swap_nbits(1, 0x55555555, dst, xtmp1, xtmp2, rtmp, vec_enc);
6057
6058 evmovdqul(xtmp1, k0, dst, true, vec_enc);
6059 vector_reverse_byte64(bt, dst, xtmp1, xtmp1, xtmp2, rtmp, vec_enc);
6060 } else {
6061 vmovdqu(xtmp1, ExternalAddress(StubRoutines::x86::vector_reverse_bit_lut()), vec_enc, rtmp);
6062 vbroadcast(T_INT, xtmp2, 0x0F0F0F0F, rtmp, vec_enc);
6063
6064 // Get the reverse bit sequence of lower nibble of each byte.
6065 vpand(dst, xtmp2, src, vec_enc);
6066 vpshufb(dst, xtmp1, dst, vec_enc);
6067 vpsllq(dst, dst, 4, vec_enc);
6068
6069 // Get the reverse bit sequence of upper nibble of each byte.
6070 vpandn(xtmp2, xtmp2, src, vec_enc);
6071 vpsrlq(xtmp2, xtmp2, 4, vec_enc);
6072 vpshufb(xtmp2, xtmp1, xtmp2, vec_enc);
6073
6074 // Perform logical OR operation b/w left shifted reverse bit sequence of lower nibble and
6075 // right shifted reverse bit sequence of upper nibble to obtain the reverse bit sequence of each byte.
6076 vpor(xtmp2, dst, xtmp2, vec_enc);
6077 vector_reverse_byte(bt, dst, xtmp2, vec_enc);
6078 }
6079 }
6080
6081 void C2_MacroAssembler::vector_reverse_bit_gfni(BasicType bt, XMMRegister dst, XMMRegister src, AddressLiteral mask, int vec_enc,
6082 XMMRegister xtmp, Register rscratch) {
6083 assert(VM_Version::supports_gfni(), "");
6084 assert(rscratch != noreg || always_reachable(mask), "missing");
6085
6086 // Galois field instruction based bit reversal based on following algorithm.
6087 // http://0x80.pl/articles/avx512-galois-field-for-bit-shuffling.html
6088 vpbroadcastq(xtmp, mask, vec_enc, rscratch);
6089 vgf2p8affineqb(xtmp, src, xtmp, 0, vec_enc);
6090 vector_reverse_byte(bt, dst, xtmp, vec_enc);
6091 }
6092
6093 void C2_MacroAssembler::vector_swap_nbits(int nbits, int bitmask, XMMRegister dst, XMMRegister src,
6094 XMMRegister xtmp1, Register rtmp, int vec_enc) {
6095 vbroadcast(T_INT, xtmp1, bitmask, rtmp, vec_enc);
6096 evpandq(dst, xtmp1, src, vec_enc);
6097 vpsllq(dst, dst, nbits, vec_enc);
6098 vpandn(xtmp1, xtmp1, src, vec_enc);
6099 vpsrlq(xtmp1, xtmp1, nbits, vec_enc);
6100 evporq(dst, dst, xtmp1, vec_enc);
6101 }
6102
6103 void C2_MacroAssembler::vector_reverse_byte64(BasicType bt, XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
6104 XMMRegister xtmp2, Register rtmp, int vec_enc) {
6105 // Shift based bit reversal.
6106 assert(VM_Version::supports_evex(), "");
6107 switch(bt) {
6108 case T_LONG:
6109 // Swap upper and lower double word of each quad word.
6110 evprorq(xtmp1, k0, src, 32, true, vec_enc);
6111 evprord(xtmp1, k0, xtmp1, 16, true, vec_enc);
6112 vector_swap_nbits(8, 0x00FF00FF, dst, xtmp1, xtmp2, rtmp, vec_enc);
6113 break;
6114 case T_INT:
6115 // Swap upper and lower word of each double word.
6116 evprord(xtmp1, k0, src, 16, true, vec_enc);
6117 vector_swap_nbits(8, 0x00FF00FF, dst, xtmp1, xtmp2, rtmp, vec_enc);
6118 break;
6119 case T_CHAR:
6120 case T_SHORT:
6121 // Swap upper and lower byte of each word.
6122 vector_swap_nbits(8, 0x00FF00FF, dst, src, xtmp2, rtmp, vec_enc);
6123 break;
6124 case T_BYTE:
6125 evmovdquq(dst, k0, src, true, vec_enc);
6126 break;
6127 default:
6128 fatal("Unsupported type %s", type2name(bt));
6129 break;
6130 }
6131 }
6132
6133 void C2_MacroAssembler::vector_reverse_byte(BasicType bt, XMMRegister dst, XMMRegister src, int vec_enc) {
6134 if (bt == T_BYTE) {
6135 if (VM_Version::supports_avx512vl() || vec_enc == Assembler::AVX_512bit) {
6136 evmovdquq(dst, k0, src, true, vec_enc);
6137 } else {
6138 vmovdqu(dst, src);
6139 }
6140 return;
6141 }
6142 // Perform byte reversal by shuffling the bytes of a multi-byte primitive type using
6143 // pre-computed shuffle indices.
6144 switch(bt) {
6145 case T_LONG:
6146 vmovdqu(dst, ExternalAddress(StubRoutines::x86::vector_reverse_byte_perm_mask_long()), vec_enc, noreg);
6147 break;
6148 case T_INT:
6149 vmovdqu(dst, ExternalAddress(StubRoutines::x86::vector_reverse_byte_perm_mask_int()), vec_enc, noreg);
6150 break;
6151 case T_CHAR:
6152 case T_SHORT:
6153 vmovdqu(dst, ExternalAddress(StubRoutines::x86::vector_reverse_byte_perm_mask_short()), vec_enc, noreg);
6154 break;
6155 default:
6156 fatal("Unsupported type %s", type2name(bt));
6157 break;
6158 }
6159 vpshufb(dst, src, dst, vec_enc);
6160 }
6161
6162 void C2_MacroAssembler::vector_count_leading_zeros_evex(BasicType bt, XMMRegister dst, XMMRegister src,
6163 XMMRegister xtmp1, XMMRegister xtmp2, XMMRegister xtmp3,
6164 KRegister ktmp, Register rtmp, bool merge, int vec_enc) {
6165 assert(is_integral_type(bt), "");
6166 assert(VM_Version::supports_avx512vl() || vec_enc == Assembler::AVX_512bit, "");
6167 assert(VM_Version::supports_avx512cd(), "");
6168 switch(bt) {
6169 case T_LONG:
6170 evplzcntq(dst, ktmp, src, merge, vec_enc);
6171 break;
6172 case T_INT:
6173 evplzcntd(dst, ktmp, src, merge, vec_enc);
6174 break;
6175 case T_SHORT:
6176 vpternlogd(xtmp1, 0xff, xtmp1, xtmp1, vec_enc);
6177 vpunpcklwd(xtmp2, xtmp1, src, vec_enc);
6178 evplzcntd(xtmp2, ktmp, xtmp2, merge, vec_enc);
6179 vpunpckhwd(dst, xtmp1, src, vec_enc);
6180 evplzcntd(dst, ktmp, dst, merge, vec_enc);
6181 vpackusdw(dst, xtmp2, dst, vec_enc);
6182 break;
6183 case T_BYTE:
6184 // T1 = Compute leading zero counts of 4 LSB bits of each byte by
6185 // accessing the lookup table.
6186 // T2 = Compute leading zero counts of 4 MSB bits of each byte by
6187 // accessing the lookup table.
6188 // Add T1 to T2 if 4 MSB bits of byte are all zeros.
6189 assert(VM_Version::supports_avx512bw(), "");
6190 evmovdquq(xtmp1, ExternalAddress(StubRoutines::x86::vector_count_leading_zeros_lut()), vec_enc, rtmp);
6191 vbroadcast(T_INT, dst, 0x0F0F0F0F, rtmp, vec_enc);
6192 vpand(xtmp2, dst, src, vec_enc);
6193 vpshufb(xtmp2, xtmp1, xtmp2, vec_enc);
6194 vpsrlw(xtmp3, src, 4, vec_enc);
6195 vpand(xtmp3, dst, xtmp3, vec_enc);
6196 vpshufb(dst, xtmp1, xtmp3, vec_enc);
6197 vpxor(xtmp1, xtmp1, xtmp1, vec_enc);
6198 evpcmpeqb(ktmp, xtmp1, xtmp3, vec_enc);
6199 evpaddb(dst, ktmp, dst, xtmp2, true, vec_enc);
6200 break;
6201 default:
6202 fatal("Unsupported type %s", type2name(bt));
6203 break;
6204 }
6205 }
6206
6207 void C2_MacroAssembler::vector_count_leading_zeros_byte_avx(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
6208 XMMRegister xtmp2, XMMRegister xtmp3, Register rtmp, int vec_enc) {
6209 vmovdqu(xtmp1, ExternalAddress(StubRoutines::x86::vector_count_leading_zeros_lut()), rtmp);
6210 vbroadcast(T_INT, xtmp2, 0x0F0F0F0F, rtmp, vec_enc);
6211 // T1 = Compute leading zero counts of 4 LSB bits of each byte by
6212 // accessing the lookup table.
6213 vpand(dst, xtmp2, src, vec_enc);
6214 vpshufb(dst, xtmp1, dst, vec_enc);
6215 // T2 = Compute leading zero counts of 4 MSB bits of each byte by
6216 // accessing the lookup table.
6217 vpsrlw(xtmp3, src, 4, vec_enc);
6218 vpand(xtmp3, xtmp2, xtmp3, vec_enc);
6219 vpshufb(xtmp2, xtmp1, xtmp3, vec_enc);
6220 // Add T1 to T2 if 4 MSB bits of byte are all zeros.
6221 vpxor(xtmp1, xtmp1, xtmp1, vec_enc);
6222 vpcmpeqb(xtmp3, xtmp1, xtmp3, vec_enc);
6223 vpaddb(dst, dst, xtmp2, vec_enc);
6224 vpblendvb(dst, xtmp2, dst, xtmp3, vec_enc);
6225 }
6226
6227 void C2_MacroAssembler::vector_count_leading_zeros_short_avx(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
6228 XMMRegister xtmp2, XMMRegister xtmp3, Register rtmp, int vec_enc) {
6229 vector_count_leading_zeros_byte_avx(dst, src, xtmp1, xtmp2, xtmp3, rtmp, vec_enc);
6230 // Add zero counts of lower byte and upper byte of a word if
6231 // upper byte holds a zero value.
6232 vpsrlw(xtmp3, src, 8, vec_enc);
6233 // xtmp1 is set to all zeros by vector_count_leading_zeros_byte_avx.
6234 vpcmpeqw(xtmp3, xtmp1, xtmp3, vec_enc);
6235 vpsllw(xtmp2, dst, 8, vec_enc);
6236 vpaddw(xtmp2, xtmp2, dst, vec_enc);
6237 vpblendvb(dst, dst, xtmp2, xtmp3, vec_enc);
6238 vpsrlw(dst, dst, 8, vec_enc);
6239 }
6240
6241 void C2_MacroAssembler::vector_count_leading_zeros_int_avx(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
6242 XMMRegister xtmp2, XMMRegister xtmp3, int vec_enc) {
6243 // By converting the integer to a float, we can obtain the number of leading zeros based on the exponent of the float.
6244 // As the float exponent contains a bias of 127 for nonzero values, the bias must be removed before interpreting the
6245 // exponent as the leading zero count.
6246
6247 // Remove the bit to the right of the highest set bit ensuring that the conversion to float cannot round up to a higher
6248 // power of 2, which has a higher exponent than the input. This transformation is valid as only the highest set bit
6249 // contributes to the leading number of zeros.
6250 vpsrld(dst, src, 1, vec_enc);
6251 vpandn(dst, dst, src, vec_enc);
6252
6253 vcvtdq2ps(dst, dst, vec_enc);
6254
6255 // By comparing the register to itself, all the bits in the destination are set.
6256 vpcmpeqd(xtmp1, xtmp1, xtmp1, vec_enc);
6257
6258 // Move the biased exponent to the low end of the lane and mask with 0xFF to discard the sign bit.
6259 vpsrld(xtmp2, xtmp1, 24, vec_enc);
6260 vpsrld(dst, dst, 23, vec_enc);
6261 vpand(dst, xtmp2, dst, vec_enc);
6262
6263 // Subtract 127 from the exponent, which removes the bias from the exponent.
6264 vpsrld(xtmp2, xtmp1, 25, vec_enc);
6265 vpsubd(dst, dst, xtmp2, vec_enc);
6266
6267 vpsrld(xtmp2, xtmp1, 27, vec_enc);
6268
6269 // If the original value is 0 the exponent would not have bias, so the subtraction creates a negative number. If this
6270 // is found in any of the lanes, replace the lane with -1 from xtmp1.
6271 vblendvps(dst, dst, xtmp1, dst, vec_enc, true, xtmp3);
6272
6273 // If the original value is negative, replace the lane with 31.
6274 vblendvps(dst, dst, xtmp2, src, vec_enc, true, xtmp3);
6275
6276 // Subtract the exponent from 31, giving the final result. For 0, the result is 32 as the exponent was replaced with -1,
6277 // and for negative numbers the result is 0 as the exponent was replaced with 31.
6278 vpsubd(dst, xtmp2, dst, vec_enc);
6279 }
6280
6281 void C2_MacroAssembler::vector_count_leading_zeros_long_avx(XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
6282 XMMRegister xtmp2, XMMRegister xtmp3, Register rtmp, int vec_enc) {
6283 // Find the leading zeros of the top and bottom halves of the long individually.
6284 vector_count_leading_zeros_int_avx(dst, src, xtmp1, xtmp2, xtmp3, vec_enc);
6285
6286 // Move the top half result to the bottom half of xtmp1, setting the top half to 0.
6287 vpsrlq(xtmp1, dst, 32, vec_enc);
6288 // By moving the top half result to the right by 6 bits, if the top half was empty (i.e. 32 is returned) the result bit will
6289 // be in the most significant position of the bottom half.
6290 vpsrlq(xtmp2, dst, 6, vec_enc);
6291
6292 // In the bottom half, add the top half and bottom half results.
6293 vpaddq(dst, xtmp1, dst, vec_enc);
6294
6295 // For the bottom half, choose between the values using the most significant bit of xtmp2.
6296 // If the MSB is set, then bottom+top in dst is the resulting value. If the top half is less than 32 xtmp1 is chosen,
6297 // which contains only the top half result.
6298 // In the top half the MSB is always zero, so the value in xtmp1 is always chosen. This value is always 0, which clears
6299 // the lane as required.
6300 vblendvps(dst, xtmp1, dst, xtmp2, vec_enc, true, xtmp3);
6301 }
6302
6303 void C2_MacroAssembler::vector_count_leading_zeros_avx(BasicType bt, XMMRegister dst, XMMRegister src,
6304 XMMRegister xtmp1, XMMRegister xtmp2, XMMRegister xtmp3,
6305 Register rtmp, int vec_enc) {
6306 assert(is_integral_type(bt), "unexpected type");
6307 assert(vec_enc < Assembler::AVX_512bit, "");
6308 switch(bt) {
6309 case T_LONG:
6310 vector_count_leading_zeros_long_avx(dst, src, xtmp1, xtmp2, xtmp3, rtmp, vec_enc);
6311 break;
6312 case T_INT:
6313 vector_count_leading_zeros_int_avx(dst, src, xtmp1, xtmp2, xtmp3, vec_enc);
6314 break;
6315 case T_SHORT:
6316 vector_count_leading_zeros_short_avx(dst, src, xtmp1, xtmp2, xtmp3, rtmp, vec_enc);
6317 break;
6318 case T_BYTE:
6319 vector_count_leading_zeros_byte_avx(dst, src, xtmp1, xtmp2, xtmp3, rtmp, vec_enc);
6320 break;
6321 default:
6322 fatal("Unsupported type %s", type2name(bt));
6323 break;
6324 }
6325 }
6326
6327 void C2_MacroAssembler::vpsub(BasicType bt, XMMRegister dst, XMMRegister src1, XMMRegister src2, int vec_enc) {
6328 switch(bt) {
6329 case T_BYTE:
6330 vpsubb(dst, src1, src2, vec_enc);
6331 break;
6332 case T_SHORT:
6333 vpsubw(dst, src1, src2, vec_enc);
6334 break;
6335 case T_INT:
6336 vpsubd(dst, src1, src2, vec_enc);
6337 break;
6338 case T_LONG:
6339 vpsubq(dst, src1, src2, vec_enc);
6340 break;
6341 default:
6342 fatal("Unsupported type %s", type2name(bt));
6343 break;
6344 }
6345 }
6346
6347 // Trailing zero count computation is based on leading zero count operation as per
6348 // following equation. All AVX3 targets support AVX512CD feature which offers
6349 // direct vector instruction to compute leading zero count.
6350 // CTZ = PRIM_TYPE_WIDHT - CLZ((x - 1) & ~x)
6351 void C2_MacroAssembler::vector_count_trailing_zeros_evex(BasicType bt, XMMRegister dst, XMMRegister src,
6352 XMMRegister xtmp1, XMMRegister xtmp2, XMMRegister xtmp3,
6353 XMMRegister xtmp4, KRegister ktmp, Register rtmp, int vec_enc) {
6354 assert(is_integral_type(bt), "");
6355 // xtmp = -1
6356 vpternlogd(xtmp4, 0xff, xtmp4, xtmp4, vec_enc);
6357 // xtmp = xtmp + src
6358 vpadd(bt, xtmp4, xtmp4, src, vec_enc);
6359 // xtmp = xtmp & ~src
6360 vpternlogd(xtmp4, 0x40, xtmp4, src, vec_enc);
6361 vector_count_leading_zeros_evex(bt, dst, xtmp4, xtmp1, xtmp2, xtmp3, ktmp, rtmp, true, vec_enc);
6362 vbroadcast(bt, xtmp4, 8 * type2aelembytes(bt), rtmp, vec_enc);
6363 vpsub(bt, dst, xtmp4, dst, vec_enc);
6364 }
6365
6366 // Trailing zero count computation for AVX2 targets is based on popcount operation as per following equation
6367 // CTZ = PRIM_TYPE_WIDHT - POPC(x | -x)
6368 void C2_MacroAssembler::vector_count_trailing_zeros_avx(BasicType bt, XMMRegister dst, XMMRegister src, XMMRegister xtmp1,
6369 XMMRegister xtmp2, XMMRegister xtmp3, Register rtmp, int vec_enc) {
6370 assert(is_integral_type(bt), "");
6371 // xtmp = 0
6372 vpxor(xtmp3 , xtmp3, xtmp3, vec_enc);
6373 // xtmp = 0 - src
6374 vpsub(bt, xtmp3, xtmp3, src, vec_enc);
6375 // xtmp = xtmp | src
6376 vpor(xtmp3, xtmp3, src, vec_enc);
6377 vector_popcount_integral(bt, dst, xtmp3, xtmp1, xtmp2, rtmp, vec_enc);
6378 vbroadcast(bt, xtmp1, 8 * type2aelembytes(bt), rtmp, vec_enc);
6379 vpsub(bt, dst, xtmp1, dst, vec_enc);
6380 }
6381
6382 void C2_MacroAssembler::udivI(Register rax, Register divisor, Register rdx) {
6383 Label done;
6384 Label neg_divisor_fastpath;
6385 cmpl(divisor, 0);
6386 jccb(Assembler::less, neg_divisor_fastpath);
6387 xorl(rdx, rdx);
6388 divl(divisor);
6389 jmpb(done);
6390 bind(neg_divisor_fastpath);
6391 // Fastpath for divisor < 0:
6392 // quotient = (dividend & ~(dividend - divisor)) >>> (Integer.SIZE - 1)
6393 // See Hacker's Delight (2nd ed), section 9.3 which is implemented in java.lang.Long.divideUnsigned()
6394 movl(rdx, rax);
6395 subl(rdx, divisor);
6396 if (VM_Version::supports_bmi1() && VM_Version::supports_avx()) {
6397 andnl(rax, rdx, rax);
6398 } else {
6399 notl(rdx);
6400 andl(rax, rdx);
6401 }
6402 shrl(rax, 31);
6403 bind(done);
6404 }
6405
6406 void C2_MacroAssembler::umodI(Register rax, Register divisor, Register rdx) {
6407 Label done;
6408 Label neg_divisor_fastpath;
6409 cmpl(divisor, 0);
6410 jccb(Assembler::less, neg_divisor_fastpath);
6411 xorl(rdx, rdx);
6412 divl(divisor);
6413 jmpb(done);
6414 bind(neg_divisor_fastpath);
6415 // Fastpath when divisor < 0:
6416 // remainder = dividend - (((dividend & ~(dividend - divisor)) >> (Integer.SIZE - 1)) & divisor)
6417 // See Hacker's Delight (2nd ed), section 9.3 which is implemented in java.lang.Long.remainderUnsigned()
6418 movl(rdx, rax);
6419 subl(rax, divisor);
6420 if (VM_Version::supports_bmi1() && VM_Version::supports_avx()) {
6421 andnl(rax, rax, rdx);
6422 } else {
6423 notl(rax);
6424 andl(rax, rdx);
6425 }
6426 sarl(rax, 31);
6427 andl(rax, divisor);
6428 subl(rdx, rax);
6429 bind(done);
6430 }
6431
6432 void C2_MacroAssembler::udivmodI(Register rax, Register divisor, Register rdx, Register tmp) {
6433 Label done;
6434 Label neg_divisor_fastpath;
6435
6436 cmpl(divisor, 0);
6437 jccb(Assembler::less, neg_divisor_fastpath);
6438 xorl(rdx, rdx);
6439 divl(divisor);
6440 jmpb(done);
6441 bind(neg_divisor_fastpath);
6442 // Fastpath for divisor < 0:
6443 // quotient = (dividend & ~(dividend - divisor)) >>> (Integer.SIZE - 1)
6444 // remainder = dividend - (((dividend & ~(dividend - divisor)) >> (Integer.SIZE - 1)) & divisor)
6445 // See Hacker's Delight (2nd ed), section 9.3 which is implemented in
6446 // java.lang.Long.divideUnsigned() and java.lang.Long.remainderUnsigned()
6447 movl(rdx, rax);
6448 subl(rax, divisor);
6449 if (VM_Version::supports_bmi1() && VM_Version::supports_avx()) {
6450 andnl(rax, rax, rdx);
6451 } else {
6452 notl(rax);
6453 andl(rax, rdx);
6454 }
6455 movl(tmp, rax);
6456 shrl(rax, 31); // quotient
6457 sarl(tmp, 31);
6458 andl(tmp, divisor);
6459 subl(rdx, tmp); // remainder
6460 bind(done);
6461 }
6462
6463 void C2_MacroAssembler::reverseI(Register dst, Register src, XMMRegister xtmp1,
6464 XMMRegister xtmp2, Register rtmp) {
6465 if(VM_Version::supports_gfni()) {
6466 // Galois field instruction based bit reversal based on following algorithm.
6467 // http://0x80.pl/articles/avx512-galois-field-for-bit-shuffling.html
6468 mov64(rtmp, 0x8040201008040201L);
6469 movq(xtmp1, src);
6470 movq(xtmp2, rtmp);
6471 gf2p8affineqb(xtmp1, xtmp2, 0);
6472 movq(dst, xtmp1);
6473 } else {
6474 // Swap even and odd numbered bits.
6475 movl(rtmp, src);
6476 andl(rtmp, 0x55555555);
6477 shll(rtmp, 1);
6478 movl(dst, src);
6479 andl(dst, 0xAAAAAAAA);
6480 shrl(dst, 1);
6481 orl(dst, rtmp);
6482
6483 // Swap LSB and MSB 2 bits of each nibble.
6484 movl(rtmp, dst);
6485 andl(rtmp, 0x33333333);
6486 shll(rtmp, 2);
6487 andl(dst, 0xCCCCCCCC);
6488 shrl(dst, 2);
6489 orl(dst, rtmp);
6490
6491 // Swap LSB and MSB 4 bits of each byte.
6492 movl(rtmp, dst);
6493 andl(rtmp, 0x0F0F0F0F);
6494 shll(rtmp, 4);
6495 andl(dst, 0xF0F0F0F0);
6496 shrl(dst, 4);
6497 orl(dst, rtmp);
6498 }
6499 bswapl(dst);
6500 }
6501
6502 void C2_MacroAssembler::reverseL(Register dst, Register src, XMMRegister xtmp1,
6503 XMMRegister xtmp2, Register rtmp1, Register rtmp2) {
6504 if(VM_Version::supports_gfni()) {
6505 // Galois field instruction based bit reversal based on following algorithm.
6506 // http://0x80.pl/articles/avx512-galois-field-for-bit-shuffling.html
6507 mov64(rtmp1, 0x8040201008040201L);
6508 movq(xtmp1, src);
6509 movq(xtmp2, rtmp1);
6510 gf2p8affineqb(xtmp1, xtmp2, 0);
6511 movq(dst, xtmp1);
6512 } else {
6513 // Swap even and odd numbered bits.
6514 movq(rtmp1, src);
6515 mov64(rtmp2, 0x5555555555555555L);
6516 andq(rtmp1, rtmp2);
6517 shlq(rtmp1, 1);
6518 movq(dst, src);
6519 notq(rtmp2);
6520 andq(dst, rtmp2);
6521 shrq(dst, 1);
6522 orq(dst, rtmp1);
6523
6524 // Swap LSB and MSB 2 bits of each nibble.
6525 movq(rtmp1, dst);
6526 mov64(rtmp2, 0x3333333333333333L);
6527 andq(rtmp1, rtmp2);
6528 shlq(rtmp1, 2);
6529 notq(rtmp2);
6530 andq(dst, rtmp2);
6531 shrq(dst, 2);
6532 orq(dst, rtmp1);
6533
6534 // Swap LSB and MSB 4 bits of each byte.
6535 movq(rtmp1, dst);
6536 mov64(rtmp2, 0x0F0F0F0F0F0F0F0FL);
6537 andq(rtmp1, rtmp2);
6538 shlq(rtmp1, 4);
6539 notq(rtmp2);
6540 andq(dst, rtmp2);
6541 shrq(dst, 4);
6542 orq(dst, rtmp1);
6543 }
6544 bswapq(dst);
6545 }
6546
6547 void C2_MacroAssembler::udivL(Register rax, Register divisor, Register rdx) {
6548 Label done;
6549 Label neg_divisor_fastpath;
6550 cmpq(divisor, 0);
6551 jccb(Assembler::less, neg_divisor_fastpath);
6552 xorl(rdx, rdx);
6553 divq(divisor);
6554 jmpb(done);
6555 bind(neg_divisor_fastpath);
6556 // Fastpath for divisor < 0:
6557 // quotient = (dividend & ~(dividend - divisor)) >>> (Long.SIZE - 1)
6558 // See Hacker's Delight (2nd ed), section 9.3 which is implemented in java.lang.Long.divideUnsigned()
6559 movq(rdx, rax);
6560 subq(rdx, divisor);
6561 if (VM_Version::supports_bmi1() && VM_Version::supports_avx()) {
6562 andnq(rax, rdx, rax);
6563 } else {
6564 notq(rdx);
6565 andq(rax, rdx);
6566 }
6567 shrq(rax, 63);
6568 bind(done);
6569 }
6570
6571 void C2_MacroAssembler::umodL(Register rax, Register divisor, Register rdx) {
6572 Label done;
6573 Label neg_divisor_fastpath;
6574 cmpq(divisor, 0);
6575 jccb(Assembler::less, neg_divisor_fastpath);
6576 xorq(rdx, rdx);
6577 divq(divisor);
6578 jmp(done);
6579 bind(neg_divisor_fastpath);
6580 // Fastpath when divisor < 0:
6581 // remainder = dividend - (((dividend & ~(dividend - divisor)) >> (Long.SIZE - 1)) & divisor)
6582 // See Hacker's Delight (2nd ed), section 9.3 which is implemented in java.lang.Long.remainderUnsigned()
6583 movq(rdx, rax);
6584 subq(rax, divisor);
6585 if (VM_Version::supports_bmi1() && VM_Version::supports_avx()) {
6586 andnq(rax, rax, rdx);
6587 } else {
6588 notq(rax);
6589 andq(rax, rdx);
6590 }
6591 sarq(rax, 63);
6592 andq(rax, divisor);
6593 subq(rdx, rax);
6594 bind(done);
6595 }
6596
6597 void C2_MacroAssembler::udivmodL(Register rax, Register divisor, Register rdx, Register tmp) {
6598 Label done;
6599 Label neg_divisor_fastpath;
6600 cmpq(divisor, 0);
6601 jccb(Assembler::less, neg_divisor_fastpath);
6602 xorq(rdx, rdx);
6603 divq(divisor);
6604 jmp(done);
6605 bind(neg_divisor_fastpath);
6606 // Fastpath for divisor < 0:
6607 // quotient = (dividend & ~(dividend - divisor)) >>> (Long.SIZE - 1)
6608 // remainder = dividend - (((dividend & ~(dividend - divisor)) >> (Long.SIZE - 1)) & divisor)
6609 // See Hacker's Delight (2nd ed), section 9.3 which is implemented in
6610 // java.lang.Long.divideUnsigned() and java.lang.Long.remainderUnsigned()
6611 movq(rdx, rax);
6612 subq(rax, divisor);
6613 if (VM_Version::supports_bmi1() && VM_Version::supports_avx()) {
6614 andnq(rax, rax, rdx);
6615 } else {
6616 notq(rax);
6617 andq(rax, rdx);
6618 }
6619 movq(tmp, rax);
6620 shrq(rax, 63); // quotient
6621 sarq(tmp, 63);
6622 andq(tmp, divisor);
6623 subq(rdx, tmp); // remainder
6624 bind(done);
6625 }
6626
6627 void C2_MacroAssembler::rearrange_bytes(XMMRegister dst, XMMRegister shuffle, XMMRegister src, XMMRegister xtmp1,
6628 XMMRegister xtmp2, XMMRegister xtmp3, Register rtmp, KRegister ktmp,
6629 int vlen_enc) {
6630 assert(VM_Version::supports_avx512bw(), "");
6631 // Byte shuffles are inlane operations and indices are determined using
6632 // lower 4 bit of each shuffle lane, thus all shuffle indices are
6633 // normalized to index range 0-15. This makes sure that all the multiples
6634 // of an index value are placed at same relative position in 128 bit
6635 // lane i.e. elements corresponding to shuffle indices 16, 32 and 64
6636 // will be 16th element in their respective 128 bit lanes.
6637 movl(rtmp, 16);
6638 evpbroadcastb(xtmp1, rtmp, vlen_enc);
6639
6640 // Compute a mask for shuffle vector by comparing indices with expression INDEX < 16,
6641 // Broadcast first 128 bit lane across entire vector, shuffle the vector lanes using
6642 // original shuffle indices and move the shuffled lanes corresponding to true
6643 // mask to destination vector.
6644 evpcmpb(ktmp, k0, shuffle, xtmp1, Assembler::lt, true, vlen_enc);
6645 evshufi64x2(xtmp2, src, src, 0x0, vlen_enc);
6646 evpshufb(dst, ktmp, xtmp2, shuffle, false, vlen_enc);
6647
6648 // Perform above steps with lane comparison expression as INDEX >= 16 && INDEX < 32
6649 // and broadcasting second 128 bit lane.
6650 evpcmpb(ktmp, k0, shuffle, xtmp1, Assembler::nlt, true, vlen_enc);
6651 vpsllq(xtmp2, xtmp1, 0x1, vlen_enc);
6652 evpcmpb(ktmp, ktmp, shuffle, xtmp2, Assembler::lt, true, vlen_enc);
6653 evshufi64x2(xtmp3, src, src, 0x55, vlen_enc);
6654 evpshufb(dst, ktmp, xtmp3, shuffle, true, vlen_enc);
6655
6656 // Perform above steps with lane comparison expression as INDEX >= 32 && INDEX < 48
6657 // and broadcasting third 128 bit lane.
6658 evpcmpb(ktmp, k0, shuffle, xtmp2, Assembler::nlt, true, vlen_enc);
6659 vpaddb(xtmp1, xtmp1, xtmp2, vlen_enc);
6660 evpcmpb(ktmp, ktmp, shuffle, xtmp1, Assembler::lt, true, vlen_enc);
6661 evshufi64x2(xtmp3, src, src, 0xAA, vlen_enc);
6662 evpshufb(dst, ktmp, xtmp3, shuffle, true, vlen_enc);
6663
6664 // Perform above steps with lane comparison expression as INDEX >= 48 && INDEX < 64
6665 // and broadcasting third 128 bit lane.
6666 evpcmpb(ktmp, k0, shuffle, xtmp1, Assembler::nlt, true, vlen_enc);
6667 vpsllq(xtmp2, xtmp2, 0x1, vlen_enc);
6668 evpcmpb(ktmp, ktmp, shuffle, xtmp2, Assembler::lt, true, vlen_enc);
6669 evshufi64x2(xtmp3, src, src, 0xFF, vlen_enc);
6670 evpshufb(dst, ktmp, xtmp3, shuffle, true, vlen_enc);
6671 }
6672
6673 void C2_MacroAssembler::vector_rearrange_int_float(BasicType bt, XMMRegister dst,
6674 XMMRegister shuffle, XMMRegister src, int vlen_enc) {
6675 if (vlen_enc == AVX_128bit) {
6676 vpermilps(dst, src, shuffle, vlen_enc);
6677 } else if (bt == T_INT) {
6678 vpermd(dst, shuffle, src, vlen_enc);
6679 } else {
6680 assert(bt == T_FLOAT, "");
6681 vpermps(dst, shuffle, src, vlen_enc);
6682 }
6683 }
6684
6685 void C2_MacroAssembler::efp16sh(int opcode, XMMRegister dst, XMMRegister src1, XMMRegister src2) {
6686 switch(opcode) {
6687 case Op_AddHF: vaddsh(dst, src1, src2); break;
6688 case Op_SubHF: vsubsh(dst, src1, src2); break;
6689 case Op_MulHF: vmulsh(dst, src1, src2); break;
6690 case Op_DivHF: vdivsh(dst, src1, src2); break;
6691 default: assert(false, "%s", NodeClassNames[opcode]); break;
6692 }
6693 }
6694
6695 void C2_MacroAssembler::vector_saturating_op(int ideal_opc, BasicType elem_bt, XMMRegister dst, XMMRegister src1, XMMRegister src2, int vlen_enc) {
6696 switch(elem_bt) {
6697 case T_BYTE:
6698 if (ideal_opc == Op_SaturatingAddV) {
6699 vpaddsb(dst, src1, src2, vlen_enc);
6700 } else {
6701 assert(ideal_opc == Op_SaturatingSubV, "");
6702 vpsubsb(dst, src1, src2, vlen_enc);
6703 }
6704 break;
6705 case T_SHORT:
6706 if (ideal_opc == Op_SaturatingAddV) {
6707 vpaddsw(dst, src1, src2, vlen_enc);
6708 } else {
6709 assert(ideal_opc == Op_SaturatingSubV, "");
6710 vpsubsw(dst, src1, src2, vlen_enc);
6711 }
6712 break;
6713 default:
6714 fatal("Unsupported type %s", type2name(elem_bt));
6715 break;
6716 }
6717 }
6718
6719 void C2_MacroAssembler::vector_saturating_unsigned_op(int ideal_opc, BasicType elem_bt, XMMRegister dst, XMMRegister src1, XMMRegister src2, int vlen_enc) {
6720 switch(elem_bt) {
6721 case T_BYTE:
6722 if (ideal_opc == Op_SaturatingAddV) {
6723 vpaddusb(dst, src1, src2, vlen_enc);
6724 } else {
6725 assert(ideal_opc == Op_SaturatingSubV, "");
6726 vpsubusb(dst, src1, src2, vlen_enc);
6727 }
6728 break;
6729 case T_SHORT:
6730 if (ideal_opc == Op_SaturatingAddV) {
6731 vpaddusw(dst, src1, src2, vlen_enc);
6732 } else {
6733 assert(ideal_opc == Op_SaturatingSubV, "");
6734 vpsubusw(dst, src1, src2, vlen_enc);
6735 }
6736 break;
6737 default:
6738 fatal("Unsupported type %s", type2name(elem_bt));
6739 break;
6740 }
6741 }
6742
6743 void C2_MacroAssembler::vector_sub_dq_saturating_unsigned_evex(BasicType elem_bt, XMMRegister dst, XMMRegister src1,
6744 XMMRegister src2, KRegister ktmp, int vlen_enc) {
6745 // For unsigned subtraction, overflow happens when magnitude of second input is greater than first input.
6746 // overflow_mask = Inp1 <u Inp2
6747 evpcmpu(elem_bt, ktmp, src2, src1, Assembler::lt, vlen_enc);
6748 // Res = overflow_mask ? Zero : INP1 - INP2 (non-commutative and non-associative)
6749 evmasked_op(elem_bt == T_INT ? Op_SubVI : Op_SubVL, elem_bt, ktmp, dst, src1, src2, false, vlen_enc, false);
6750 }
6751
6752 void C2_MacroAssembler::vector_sub_dq_saturating_unsigned_avx(BasicType elem_bt, XMMRegister dst, XMMRegister src1, XMMRegister src2,
6753 XMMRegister xtmp1, XMMRegister xtmp2, int vlen_enc) {
6754 // Emulate unsigned comparison using signed comparison
6755 // Mask = Inp1 <u Inp2 => Inp1 + MIN_VALUE < Inp2 + MIN_VALUE
6756 vpgenmin_value(elem_bt, xtmp1, xtmp1, vlen_enc, true);
6757 vpadd(elem_bt, xtmp2, src1, xtmp1, vlen_enc);
6758 vpadd(elem_bt, xtmp1, src2, xtmp1, vlen_enc);
6759
6760 vpcmpgt(elem_bt, xtmp2, xtmp1, xtmp2, vlen_enc);
6761
6762 // Res = INP1 - INP2 (non-commutative and non-associative)
6763 vpsub(elem_bt, dst, src1, src2, vlen_enc);
6764 // Res = Mask ? Zero : Res
6765 vpxor(xtmp1, xtmp1, xtmp1, vlen_enc);
6766 vpblendvb(dst, dst, xtmp1, xtmp2, vlen_enc);
6767 }
6768
6769 void C2_MacroAssembler::vector_add_dq_saturating_unsigned_evex(BasicType elem_bt, XMMRegister dst, XMMRegister src1, XMMRegister src2,
6770 XMMRegister xtmp1, XMMRegister xtmp2, KRegister ktmp, int vlen_enc) {
6771 // Unsigned values ranges comprise of only +ve numbers, thus there exist only an upper bound saturation.
6772 // overflow_mask = (SRC1 + SRC2) <u (SRC1 | SRC2)
6773 // Res = Signed Add INP1, INP2
6774 vpadd(elem_bt, dst, src1, src2, vlen_enc);
6775 // T1 = SRC1 | SRC2
6776 vpor(xtmp1, src1, src2, vlen_enc);
6777 // Max_Unsigned = -1
6778 vpternlogd(xtmp2, 0xff, xtmp2, xtmp2, vlen_enc);
6779 // Unsigned compare: Mask = Res <u T1
6780 evpcmpu(elem_bt, ktmp, dst, xtmp1, Assembler::lt, vlen_enc);
6781 // res = Mask ? Max_Unsigned : Res
6782 evpblend(elem_bt, dst, ktmp, dst, xtmp2, true, vlen_enc);
6783 }
6784
6785 //
6786 // Section 2-13 Hacker's Delight list following overflow detection check for saturating
6787 // unsigned addition operation.
6788 // overflow_mask = ((a & b) | ((a | b) & ~( a + b))) >>> 31 == 1
6789 //
6790 // We empirically determined its semantic equivalence to following reduced expression
6791 // overflow_mask = (a + b) <u (a | b)
6792 //
6793 // and also verified it though Alive2 solver.
6794 // (https://alive2.llvm.org/ce/z/XDQ7dY)
6795 //
6796
6797 void C2_MacroAssembler::vector_add_dq_saturating_unsigned_avx(BasicType elem_bt, XMMRegister dst, XMMRegister src1, XMMRegister src2,
6798 XMMRegister xtmp1, XMMRegister xtmp2, XMMRegister xtmp3, int vlen_enc) {
6799 // Res = Signed Add INP1, INP2
6800 vpadd(elem_bt, dst, src1, src2, vlen_enc);
6801 // Compute T1 = INP1 | INP2
6802 vpor(xtmp3, src1, src2, vlen_enc);
6803 // T1 = Minimum signed value.
6804 vpgenmin_value(elem_bt, xtmp2, xtmp1, vlen_enc, true);
6805 // Convert T1 to signed value, T1 = T1 + MIN_VALUE
6806 vpadd(elem_bt, xtmp3, xtmp3, xtmp2, vlen_enc);
6807 // Convert Res to signed value, Res<s> = Res + MIN_VALUE
6808 vpadd(elem_bt, xtmp2, xtmp2, dst, vlen_enc);
6809 // Compute overflow detection mask = Res<1> <s T1
6810 if (elem_bt == T_INT) {
6811 vpcmpgtd(xtmp3, xtmp3, xtmp2, vlen_enc);
6812 } else {
6813 assert(elem_bt == T_LONG, "");
6814 vpcmpgtq(xtmp3, xtmp3, xtmp2, vlen_enc);
6815 }
6816 vpblendvb(dst, dst, xtmp1, xtmp3, vlen_enc);
6817 }
6818
6819 void C2_MacroAssembler::evpmovq2m_emu(KRegister ktmp, XMMRegister src, XMMRegister xtmp1, XMMRegister xtmp2,
6820 int vlen_enc, bool xtmp2_hold_M1) {
6821 if (VM_Version::supports_avx512dq()) {
6822 evpmovq2m(ktmp, src, vlen_enc);
6823 } else {
6824 assert(VM_Version::supports_evex(), "");
6825 if (!xtmp2_hold_M1) {
6826 vpternlogq(xtmp2, 0xff, xtmp2, xtmp2, vlen_enc);
6827 }
6828 evpsraq(xtmp1, src, 63, vlen_enc);
6829 evpcmpeqq(ktmp, k0, xtmp1, xtmp2, vlen_enc);
6830 }
6831 }
6832
6833 void C2_MacroAssembler::evpmovd2m_emu(KRegister ktmp, XMMRegister src, XMMRegister xtmp1, XMMRegister xtmp2,
6834 int vlen_enc, bool xtmp2_hold_M1) {
6835 if (VM_Version::supports_avx512dq()) {
6836 evpmovd2m(ktmp, src, vlen_enc);
6837 } else {
6838 assert(VM_Version::supports_evex(), "");
6839 if (!xtmp2_hold_M1) {
6840 vpternlogd(xtmp2, 0xff, xtmp2, xtmp2, vlen_enc);
6841 }
6842 vpsrad(xtmp1, src, 31, vlen_enc);
6843 Assembler::evpcmpeqd(ktmp, k0, xtmp1, xtmp2, vlen_enc);
6844 }
6845 }
6846
6847
6848 void C2_MacroAssembler::vpsign_extend_dq(BasicType elem_bt, XMMRegister dst, XMMRegister src, int vlen_enc) {
6849 if (elem_bt == T_LONG) {
6850 if (VM_Version::supports_evex()) {
6851 evpsraq(dst, src, 63, vlen_enc);
6852 } else {
6853 vpsrad(dst, src, 31, vlen_enc);
6854 vpshufd(dst, dst, 0xF5, vlen_enc);
6855 }
6856 } else {
6857 assert(elem_bt == T_INT, "");
6858 vpsrad(dst, src, 31, vlen_enc);
6859 }
6860 }
6861
6862 void C2_MacroAssembler::vpgenmax_value(BasicType elem_bt, XMMRegister dst, XMMRegister allones, int vlen_enc, bool compute_allones) {
6863 if (compute_allones) {
6864 if (VM_Version::supports_avx512vl() || vlen_enc == Assembler::AVX_512bit) {
6865 vpternlogd(allones, 0xff, allones, allones, vlen_enc);
6866 } else {
6867 vpcmpeqq(allones, allones, allones, vlen_enc);
6868 }
6869 }
6870 if (elem_bt == T_LONG) {
6871 vpsrlq(dst, allones, 1, vlen_enc);
6872 } else {
6873 assert(elem_bt == T_INT, "");
6874 vpsrld(dst, allones, 1, vlen_enc);
6875 }
6876 }
6877
6878 void C2_MacroAssembler::vpgenmin_value(BasicType elem_bt, XMMRegister dst, XMMRegister allones, int vlen_enc, bool compute_allones) {
6879 if (compute_allones) {
6880 if (VM_Version::supports_avx512vl() || vlen_enc == Assembler::AVX_512bit) {
6881 vpternlogd(allones, 0xff, allones, allones, vlen_enc);
6882 } else {
6883 vpcmpeqq(allones, allones, allones, vlen_enc);
6884 }
6885 }
6886 if (elem_bt == T_LONG) {
6887 vpsllq(dst, allones, 63, vlen_enc);
6888 } else {
6889 assert(elem_bt == T_INT, "");
6890 vpslld(dst, allones, 31, vlen_enc);
6891 }
6892 }
6893
6894 void C2_MacroAssembler::evpcmpu(BasicType elem_bt, KRegister kmask, XMMRegister src1, XMMRegister src2,
6895 Assembler::ComparisonPredicate cond, int vlen_enc) {
6896 switch(elem_bt) {
6897 case T_LONG: evpcmpuq(kmask, src1, src2, cond, vlen_enc); break;
6898 case T_INT: evpcmpud(kmask, src1, src2, cond, vlen_enc); break;
6899 case T_SHORT: evpcmpuw(kmask, src1, src2, cond, vlen_enc); break;
6900 case T_BYTE: evpcmpub(kmask, src1, src2, cond, vlen_enc); break;
6901 default: fatal("Unsupported type %s", type2name(elem_bt)); break;
6902 }
6903 }
6904
6905 void C2_MacroAssembler::vpcmpgt(BasicType elem_bt, XMMRegister dst, XMMRegister src1, XMMRegister src2, int vlen_enc) {
6906 switch(elem_bt) {
6907 case T_LONG: vpcmpgtq(dst, src1, src2, vlen_enc); break;
6908 case T_INT: vpcmpgtd(dst, src1, src2, vlen_enc); break;
6909 case T_SHORT: vpcmpgtw(dst, src1, src2, vlen_enc); break;
6910 case T_BYTE: vpcmpgtb(dst, src1, src2, vlen_enc); break;
6911 default: fatal("Unsupported type %s", type2name(elem_bt)); break;
6912 }
6913 }
6914
6915 void C2_MacroAssembler::evpmov_vec_to_mask(BasicType elem_bt, KRegister ktmp, XMMRegister src, XMMRegister xtmp1,
6916 XMMRegister xtmp2, int vlen_enc, bool xtmp2_hold_M1) {
6917 if (elem_bt == T_LONG) {
6918 evpmovq2m_emu(ktmp, src, xtmp1, xtmp2, vlen_enc, xtmp2_hold_M1);
6919 } else {
6920 assert(elem_bt == T_INT, "");
6921 evpmovd2m_emu(ktmp, src, xtmp1, xtmp2, vlen_enc, xtmp2_hold_M1);
6922 }
6923 }
6924
6925 void C2_MacroAssembler::vector_addsub_dq_saturating_evex(int ideal_opc, BasicType elem_bt, XMMRegister dst, XMMRegister src1,
6926 XMMRegister src2, XMMRegister xtmp1, XMMRegister xtmp2,
6927 KRegister ktmp1, KRegister ktmp2, int vlen_enc) {
6928 assert(elem_bt == T_INT || elem_bt == T_LONG, "");
6929 // Addition/Subtraction happens over two's compliment representation of numbers and is agnostic to signed'ness.
6930 // Overflow detection based on Hacker's delight section 2-13.
6931 if (ideal_opc == Op_SaturatingAddV) {
6932 // res = src1 + src2
6933 vpadd(elem_bt, dst, src1, src2, vlen_enc);
6934 // Overflow occurs if result polarity does not comply with equivalent polarity inputs.
6935 // overflow = (((res ^ src1) & (res ^ src2)) >>> 31(I)/63(L)) == 1
6936 vpxor(xtmp1, dst, src1, vlen_enc);
6937 vpxor(xtmp2, dst, src2, vlen_enc);
6938 vpand(xtmp2, xtmp1, xtmp2, vlen_enc);
6939 } else {
6940 assert(ideal_opc == Op_SaturatingSubV, "");
6941 // res = src1 - src2
6942 vpsub(elem_bt, dst, src1, src2, vlen_enc);
6943 // Overflow occurs when both inputs have opposite polarity and
6944 // result polarity does not comply with first input polarity.
6945 // overflow = ((src1 ^ src2) & (res ^ src1) >>> 31(I)/63(L)) == 1;
6946 vpxor(xtmp1, src1, src2, vlen_enc);
6947 vpxor(xtmp2, dst, src1, vlen_enc);
6948 vpand(xtmp2, xtmp1, xtmp2, vlen_enc);
6949 }
6950
6951 // Compute overflow detection mask.
6952 evpmov_vec_to_mask(elem_bt, ktmp1, xtmp2, xtmp2, xtmp1, vlen_enc);
6953 // Note: xtmp1 hold -1 in all its lanes after above call.
6954
6955 // Compute mask based on first input polarity.
6956 evpmov_vec_to_mask(elem_bt, ktmp2, src1, xtmp2, xtmp1, vlen_enc, true);
6957
6958 vpgenmax_value(elem_bt, xtmp2, xtmp1, vlen_enc, true);
6959 vpgenmin_value(elem_bt, xtmp1, xtmp1, vlen_enc);
6960
6961 // Compose a vector of saturating (MAX/MIN) values, where lanes corresponding to
6962 // set bits in first input polarity mask holds a min value.
6963 evpblend(elem_bt, xtmp2, ktmp2, xtmp2, xtmp1, true, vlen_enc);
6964 // Blend destination lanes with saturated values using overflow detection mask.
6965 evpblend(elem_bt, dst, ktmp1, dst, xtmp2, true, vlen_enc);
6966 }
6967
6968
6969 void C2_MacroAssembler::vector_addsub_dq_saturating_avx(int ideal_opc, BasicType elem_bt, XMMRegister dst, XMMRegister src1,
6970 XMMRegister src2, XMMRegister xtmp1, XMMRegister xtmp2,
6971 XMMRegister xtmp3, XMMRegister xtmp4, int vlen_enc) {
6972 assert(elem_bt == T_INT || elem_bt == T_LONG, "");
6973 // Addition/Subtraction happens over two's compliment representation of numbers and is agnostic to signed'ness.
6974 // Overflow detection based on Hacker's delight section 2-13.
6975 if (ideal_opc == Op_SaturatingAddV) {
6976 // res = src1 + src2
6977 vpadd(elem_bt, dst, src1, src2, vlen_enc);
6978 // Overflow occurs if result polarity does not comply with equivalent polarity inputs.
6979 // overflow = (((res ^ src1) & (res ^ src2)) >>> 31(I)/63(L)) == 1
6980 vpxor(xtmp1, dst, src1, vlen_enc);
6981 vpxor(xtmp2, dst, src2, vlen_enc);
6982 vpand(xtmp2, xtmp1, xtmp2, vlen_enc);
6983 } else {
6984 assert(ideal_opc == Op_SaturatingSubV, "");
6985 // res = src1 - src2
6986 vpsub(elem_bt, dst, src1, src2, vlen_enc);
6987 // Overflow occurs when both inputs have opposite polarity and
6988 // result polarity does not comply with first input polarity.
6989 // overflow = ((src1 ^ src2) & (res ^ src1) >>> 31(I)/63(L)) == 1;
6990 vpxor(xtmp1, src1, src2, vlen_enc);
6991 vpxor(xtmp2, dst, src1, vlen_enc);
6992 vpand(xtmp2, xtmp1, xtmp2, vlen_enc);
6993 }
6994
6995 // Sign-extend to compute overflow detection mask.
6996 vpsign_extend_dq(elem_bt, xtmp3, xtmp2, vlen_enc);
6997
6998 vpcmpeqd(xtmp1, xtmp1, xtmp1, vlen_enc);
6999 vpgenmax_value(elem_bt, xtmp2, xtmp1, vlen_enc);
7000 vpgenmin_value(elem_bt, xtmp1, xtmp1, vlen_enc);
7001
7002 // Compose saturating min/max vector using first input polarity mask.
7003 vpsign_extend_dq(elem_bt, xtmp4, src1, vlen_enc);
7004 vpblendvb(xtmp1, xtmp2, xtmp1, xtmp4, vlen_enc);
7005
7006 // Blend result with saturating vector using overflow detection mask.
7007 vpblendvb(dst, dst, xtmp1, xtmp3, vlen_enc);
7008 }
7009
7010 void C2_MacroAssembler::vector_saturating_op(int ideal_opc, BasicType elem_bt, XMMRegister dst, XMMRegister src1, Address src2, int vlen_enc) {
7011 switch(elem_bt) {
7012 case T_BYTE:
7013 if (ideal_opc == Op_SaturatingAddV) {
7014 vpaddsb(dst, src1, src2, vlen_enc);
7015 } else {
7016 assert(ideal_opc == Op_SaturatingSubV, "");
7017 vpsubsb(dst, src1, src2, vlen_enc);
7018 }
7019 break;
7020 case T_SHORT:
7021 if (ideal_opc == Op_SaturatingAddV) {
7022 vpaddsw(dst, src1, src2, vlen_enc);
7023 } else {
7024 assert(ideal_opc == Op_SaturatingSubV, "");
7025 vpsubsw(dst, src1, src2, vlen_enc);
7026 }
7027 break;
7028 default:
7029 fatal("Unsupported type %s", type2name(elem_bt));
7030 break;
7031 }
7032 }
7033
7034 void C2_MacroAssembler::vector_saturating_unsigned_op(int ideal_opc, BasicType elem_bt, XMMRegister dst, XMMRegister src1, Address src2, int vlen_enc) {
7035 switch(elem_bt) {
7036 case T_BYTE:
7037 if (ideal_opc == Op_SaturatingAddV) {
7038 vpaddusb(dst, src1, src2, vlen_enc);
7039 } else {
7040 assert(ideal_opc == Op_SaturatingSubV, "");
7041 vpsubusb(dst, src1, src2, vlen_enc);
7042 }
7043 break;
7044 case T_SHORT:
7045 if (ideal_opc == Op_SaturatingAddV) {
7046 vpaddusw(dst, src1, src2, vlen_enc);
7047 } else {
7048 assert(ideal_opc == Op_SaturatingSubV, "");
7049 vpsubusw(dst, src1, src2, vlen_enc);
7050 }
7051 break;
7052 default:
7053 fatal("Unsupported type %s", type2name(elem_bt));
7054 break;
7055 }
7056 }
7057
7058 void C2_MacroAssembler::select_from_two_vectors_evex(BasicType elem_bt, XMMRegister dst, XMMRegister src1,
7059 XMMRegister src2, int vlen_enc) {
7060 switch(elem_bt) {
7061 case T_BYTE:
7062 evpermi2b(dst, src1, src2, vlen_enc);
7063 break;
7064 case T_SHORT:
7065 evpermi2w(dst, src1, src2, vlen_enc);
7066 break;
7067 case T_INT:
7068 evpermi2d(dst, src1, src2, vlen_enc);
7069 break;
7070 case T_LONG:
7071 evpermi2q(dst, src1, src2, vlen_enc);
7072 break;
7073 case T_FLOAT:
7074 evpermi2ps(dst, src1, src2, vlen_enc);
7075 break;
7076 case T_DOUBLE:
7077 evpermi2pd(dst, src1, src2, vlen_enc);
7078 break;
7079 default:
7080 fatal("Unsupported type %s", type2name(elem_bt));
7081 break;
7082 }
7083 }
7084
7085 void C2_MacroAssembler::vector_saturating_op(int ideal_opc, BasicType elem_bt, XMMRegister dst, XMMRegister src1, XMMRegister src2, bool is_unsigned, int vlen_enc) {
7086 if (is_unsigned) {
7087 vector_saturating_unsigned_op(ideal_opc, elem_bt, dst, src1, src2, vlen_enc);
7088 } else {
7089 vector_saturating_op(ideal_opc, elem_bt, dst, src1, src2, vlen_enc);
7090 }
7091 }
7092
7093 void C2_MacroAssembler::vector_saturating_op(int ideal_opc, BasicType elem_bt, XMMRegister dst, XMMRegister src1, Address src2, bool is_unsigned, int vlen_enc) {
7094 if (is_unsigned) {
7095 vector_saturating_unsigned_op(ideal_opc, elem_bt, dst, src1, src2, vlen_enc);
7096 } else {
7097 vector_saturating_op(ideal_opc, elem_bt, dst, src1, src2, vlen_enc);
7098 }
7099 }
7100
7101 void C2_MacroAssembler::evfp16ph(int opcode, XMMRegister dst, XMMRegister src1, XMMRegister src2, int vlen_enc) {
7102 switch(opcode) {
7103 case Op_AddVHF: evaddph(dst, src1, src2, vlen_enc); break;
7104 case Op_SubVHF: evsubph(dst, src1, src2, vlen_enc); break;
7105 case Op_MulVHF: evmulph(dst, src1, src2, vlen_enc); break;
7106 case Op_DivVHF: evdivph(dst, src1, src2, vlen_enc); break;
7107 default: assert(false, "%s", NodeClassNames[opcode]); break;
7108 }
7109 }
7110
7111 void C2_MacroAssembler::evfp16ph(int opcode, XMMRegister dst, XMMRegister src1, Address src2, int vlen_enc) {
7112 switch(opcode) {
7113 case Op_AddVHF: evaddph(dst, src1, src2, vlen_enc); break;
7114 case Op_SubVHF: evsubph(dst, src1, src2, vlen_enc); break;
7115 case Op_MulVHF: evmulph(dst, src1, src2, vlen_enc); break;
7116 case Op_DivVHF: evdivph(dst, src1, src2, vlen_enc); break;
7117 default: assert(false, "%s", NodeClassNames[opcode]); break;
7118 }
7119 }
7120
7121 void C2_MacroAssembler::sminmax_fp16(int opcode, XMMRegister dst, XMMRegister src1, XMMRegister src2,
7122 KRegister ktmp, XMMRegister xtmp1, XMMRegister xtmp2) {
7123 vminmax_fp16(opcode, dst, src1, src2, ktmp, xtmp1, xtmp2, Assembler::AVX_128bit);
7124 }
7125
7126 void C2_MacroAssembler::sminmax_fp16_avx10_2(int opcode, XMMRegister dst, XMMRegister src1, XMMRegister src2,
7127 KRegister ktmp) {
7128 if (opcode == Op_MaxHF) {
7129 // dst = max(src1, src2)
7130 evminmaxsh(dst, ktmp, src1, src2, true, AVX10_2_MINMAX_MAX_COMPARE_SIGN);
7131 } else {
7132 assert(opcode == Op_MinHF, "");
7133 // dst = min(src1, src2)
7134 evminmaxsh(dst, ktmp, src1, src2, true, AVX10_2_MINMAX_MIN_COMPARE_SIGN);
7135 }
7136 }
7137
7138 void C2_MacroAssembler::vminmax_fp16(int opcode, XMMRegister dst, XMMRegister src1, XMMRegister src2,
7139 KRegister ktmp, XMMRegister xtmp1, XMMRegister xtmp2, int vlen_enc) {
7140 if (opcode == Op_MaxVHF || opcode == Op_MaxHF) {
7141 // Move sign bits of src2 to mask register.
7142 evpmovw2m(ktmp, src2, vlen_enc);
7143 // xtmp1 = src2 < 0 ? src2 : src1
7144 evpblendmw(xtmp1, ktmp, src1, src2, true, vlen_enc);
7145 // xtmp2 = src2 < 0 ? ? src1 : src2
7146 evpblendmw(xtmp2, ktmp, src2, src1, true, vlen_enc);
7147 // Idea behind above swapping is to make seconds source operand a +ve value.
7148 // As per instruction semantic, if the values being compared are both 0.0s (of either sign), the value in
7149 // the second source operand is returned. If only one value is a NaN (SNaN or QNaN) for this instruction,
7150 // the second source operand, either a NaN or a valid floating-point value, is returned
7151 // dst = max(xtmp1, xtmp2)
7152 evmaxph(dst, xtmp1, xtmp2, vlen_enc);
7153 // isNaN = is_unordered_quiet(xtmp1)
7154 evcmpph(ktmp, k0, xtmp1, xtmp1, Assembler::UNORD_Q, vlen_enc);
7155 // Final result is same as first source if its a NaN value,
7156 // in case second operand holds a NaN value then as per above semantics
7157 // result is same as second operand.
7158 Assembler::evmovdquw(dst, ktmp, xtmp1, true, vlen_enc);
7159 } else {
7160 assert(opcode == Op_MinVHF || opcode == Op_MinHF, "");
7161 // Move sign bits of src1 to mask register.
7162 evpmovw2m(ktmp, src1, vlen_enc);
7163 // xtmp1 = src1 < 0 ? src2 : src1
7164 evpblendmw(xtmp1, ktmp, src1, src2, true, vlen_enc);
7165 // xtmp2 = src1 < 0 ? src1 : src2
7166 evpblendmw(xtmp2, ktmp, src2, src1, true, vlen_enc);
7167 // Idea behind above swapping is to make seconds source operand a -ve value.
7168 // As per instruction semantics, if the values being compared are both 0.0s (of either sign), the value in
7169 // the second source operand is returned.
7170 // If only one value is a NaN (SNaN or QNaN) for this instruction, the second source operand, either a NaN
7171 // or a valid floating-point value, is written to the result.
7172 // dst = min(xtmp1, xtmp2)
7173 evminph(dst, xtmp1, xtmp2, vlen_enc);
7174 // isNaN = is_unordered_quiet(xtmp1)
7175 evcmpph(ktmp, k0, xtmp1, xtmp1, Assembler::UNORD_Q, vlen_enc);
7176 // Final result is same as first source if its a NaN value,
7177 // in case second operand holds a NaN value then as per above semantics
7178 // result is same as second operand.
7179 Assembler::evmovdquw(dst, ktmp, xtmp1, true, vlen_enc);
7180 }
7181 }
7182
7183 void C2_MacroAssembler::vminmax_fp16_avx10_2(int opcode, XMMRegister dst, XMMRegister src1, XMMRegister src2,
7184 KRegister ktmp, int vlen_enc) {
7185 if (opcode == Op_MaxVHF) {
7186 // dst = max(src1, src2)
7187 evminmaxph(dst, ktmp, src1, src2, true, AVX10_2_MINMAX_MAX_COMPARE_SIGN, vlen_enc);
7188 } else {
7189 assert(opcode == Op_MinVHF, "");
7190 // dst = min(src1, src2)
7191 evminmaxph(dst, ktmp, src1, src2, true, AVX10_2_MINMAX_MIN_COMPARE_SIGN, vlen_enc);
7192 }
7193 }
7194
7195 void C2_MacroAssembler::vminmax_fp16_avx10_2(int opcode, XMMRegister dst, XMMRegister src1, Address src2,
7196 KRegister ktmp, int vlen_enc) {
7197 if (opcode == Op_MaxVHF) {
7198 // dst = max(src1, src2)
7199 evminmaxph(dst, ktmp, src1, src2, true, AVX10_2_MINMAX_MAX_COMPARE_SIGN, vlen_enc);
7200 } else {
7201 assert(opcode == Op_MinVHF, "");
7202 // dst = min(src1, src2)
7203 evminmaxph(dst, ktmp, src1, src2, true, AVX10_2_MINMAX_MIN_COMPARE_SIGN, vlen_enc);
7204 }
7205 }
7206
7207 int C2_MacroAssembler::vector_iota_entry_index(BasicType bt) {
7208 // The vector iota entries array is ordered by type B/S/I/L/F/D, and
7209 // the offset between two types is 16.
7210 switch(bt) {
7211 case T_BYTE:
7212 return 0;
7213 case T_SHORT:
7214 return 1;
7215 case T_INT:
7216 return 2;
7217 case T_LONG:
7218 return 3;
7219 case T_FLOAT:
7220 return 4;
7221 case T_DOUBLE:
7222 return 5;
7223 default:
7224 ShouldNotReachHere();
7225 }
7226 }