162306a36Sopenharmony_ci/* SPDX-License-Identifier: GPL-2.0 */ 262306a36Sopenharmony_ci/* Written 2003 by Andi Kleen, based on a kernel by Evandro Menezes */ 362306a36Sopenharmony_ci 462306a36Sopenharmony_ci#include <linux/linkage.h> 562306a36Sopenharmony_ci#include <asm/cpufeatures.h> 662306a36Sopenharmony_ci#include <asm/alternative.h> 762306a36Sopenharmony_ci#include <asm/export.h> 862306a36Sopenharmony_ci 962306a36Sopenharmony_ci/* 1062306a36Sopenharmony_ci * Some CPUs run faster using the string copy instructions (sane microcode). 1162306a36Sopenharmony_ci * It is also a lot simpler. Use this when possible. But, don't use streaming 1262306a36Sopenharmony_ci * copy unless the CPU indicates X86_FEATURE_REP_GOOD. Could vary the 1362306a36Sopenharmony_ci * prefetch distance based on SMP/UP. 1462306a36Sopenharmony_ci */ 1562306a36Sopenharmony_ci ALIGN 1662306a36Sopenharmony_ciSYM_FUNC_START(copy_page) 1762306a36Sopenharmony_ci ALTERNATIVE "jmp copy_page_regs", "", X86_FEATURE_REP_GOOD 1862306a36Sopenharmony_ci movl $4096/8, %ecx 1962306a36Sopenharmony_ci rep movsq 2062306a36Sopenharmony_ci RET 2162306a36Sopenharmony_ciSYM_FUNC_END(copy_page) 2262306a36Sopenharmony_ciEXPORT_SYMBOL(copy_page) 2362306a36Sopenharmony_ci 2462306a36Sopenharmony_ciSYM_FUNC_START_LOCAL(copy_page_regs) 2562306a36Sopenharmony_ci subq $2*8, %rsp 2662306a36Sopenharmony_ci movq %rbx, (%rsp) 2762306a36Sopenharmony_ci movq %r12, 1*8(%rsp) 2862306a36Sopenharmony_ci 2962306a36Sopenharmony_ci movl $(4096/64)-5, %ecx 3062306a36Sopenharmony_ci .p2align 4 3162306a36Sopenharmony_ci.Loop64: 3262306a36Sopenharmony_ci dec %rcx 3362306a36Sopenharmony_ci movq 0x8*0(%rsi), %rax 3462306a36Sopenharmony_ci movq 0x8*1(%rsi), %rbx 3562306a36Sopenharmony_ci movq 0x8*2(%rsi), %rdx 3662306a36Sopenharmony_ci movq 0x8*3(%rsi), %r8 3762306a36Sopenharmony_ci movq 0x8*4(%rsi), %r9 3862306a36Sopenharmony_ci movq 0x8*5(%rsi), %r10 3962306a36Sopenharmony_ci movq 0x8*6(%rsi), %r11 4062306a36Sopenharmony_ci movq 0x8*7(%rsi), %r12 4162306a36Sopenharmony_ci 4262306a36Sopenharmony_ci prefetcht0 5*64(%rsi) 4362306a36Sopenharmony_ci 4462306a36Sopenharmony_ci movq %rax, 0x8*0(%rdi) 4562306a36Sopenharmony_ci movq %rbx, 0x8*1(%rdi) 4662306a36Sopenharmony_ci movq %rdx, 0x8*2(%rdi) 4762306a36Sopenharmony_ci movq %r8, 0x8*3(%rdi) 4862306a36Sopenharmony_ci movq %r9, 0x8*4(%rdi) 4962306a36Sopenharmony_ci movq %r10, 0x8*5(%rdi) 5062306a36Sopenharmony_ci movq %r11, 0x8*6(%rdi) 5162306a36Sopenharmony_ci movq %r12, 0x8*7(%rdi) 5262306a36Sopenharmony_ci 5362306a36Sopenharmony_ci leaq 64 (%rsi), %rsi 5462306a36Sopenharmony_ci leaq 64 (%rdi), %rdi 5562306a36Sopenharmony_ci 5662306a36Sopenharmony_ci jnz .Loop64 5762306a36Sopenharmony_ci 5862306a36Sopenharmony_ci movl $5, %ecx 5962306a36Sopenharmony_ci .p2align 4 6062306a36Sopenharmony_ci.Loop2: 6162306a36Sopenharmony_ci decl %ecx 6262306a36Sopenharmony_ci 6362306a36Sopenharmony_ci movq 0x8*0(%rsi), %rax 6462306a36Sopenharmony_ci movq 0x8*1(%rsi), %rbx 6562306a36Sopenharmony_ci movq 0x8*2(%rsi), %rdx 6662306a36Sopenharmony_ci movq 0x8*3(%rsi), %r8 6762306a36Sopenharmony_ci movq 0x8*4(%rsi), %r9 6862306a36Sopenharmony_ci movq 0x8*5(%rsi), %r10 6962306a36Sopenharmony_ci movq 0x8*6(%rsi), %r11 7062306a36Sopenharmony_ci movq 0x8*7(%rsi), %r12 7162306a36Sopenharmony_ci 7262306a36Sopenharmony_ci movq %rax, 0x8*0(%rdi) 7362306a36Sopenharmony_ci movq %rbx, 0x8*1(%rdi) 7462306a36Sopenharmony_ci movq %rdx, 0x8*2(%rdi) 7562306a36Sopenharmony_ci movq %r8, 0x8*3(%rdi) 7662306a36Sopenharmony_ci movq %r9, 0x8*4(%rdi) 7762306a36Sopenharmony_ci movq %r10, 0x8*5(%rdi) 7862306a36Sopenharmony_ci movq %r11, 0x8*6(%rdi) 7962306a36Sopenharmony_ci movq %r12, 0x8*7(%rdi) 8062306a36Sopenharmony_ci 8162306a36Sopenharmony_ci leaq 64(%rdi), %rdi 8262306a36Sopenharmony_ci leaq 64(%rsi), %rsi 8362306a36Sopenharmony_ci jnz .Loop2 8462306a36Sopenharmony_ci 8562306a36Sopenharmony_ci movq (%rsp), %rbx 8662306a36Sopenharmony_ci movq 1*8(%rsp), %r12 8762306a36Sopenharmony_ci addq $2*8, %rsp 8862306a36Sopenharmony_ci RET 8962306a36Sopenharmony_ciSYM_FUNC_END(copy_page_regs) 90