I am testing not on an Orange Pi RV2 but on a Bit-Brick K1 which has a genuine K1 (as far as I can tell).
Using time gave ~ 19s, consistent with the timing I got using rdtime (so that confirms it's apparently not just a clock issue).
Here is the disassembly of the test code:
.file "primes.c"
.option nopic
.attribute arch, "rv64i2p1_m2p0_a2p1_f2p2_d2p2_c2p0_zicbom1p0_zicbop1p0_zicboz1p0_zicond_zicsr2p0_zifencei2p0_zba1p0_zbb1p0_zbc1p0_zbs1p0"
.attribute unaligned_access, 1
.attribute stack_align, 16
.text
.section .rodata.str1.8,"aMS",@progbits,1
.align 3
.LC0:
.string "SpacemiT X60: countPrimes() ..."
.align 3
.LC1:
.string "nPrimes = %d\n"
.align 3
.LC2:
.string "Clock cycles = %g\n"
.section .text.startup,"ax",@progbits
.align 1
.globl main
.type main, [member=46715]function[/member]
main:
.LFB55:
.cfi_startproc
addi sp,sp,-32
.cfi_def_cfa_offset 32
lui a0,%hi(.LC0)
addi a0,a0,%lo(.LC0)
sd ra,24(sp)
sd s0,16(sp)
sd s1,8(sp)
.cfi_offset 1, -8
.cfi_offset 8, -16
.cfi_offset 9, -24
call puts
#APP
# 19 "primes.c" 1
rdtime s1
# 0 "" 2
#NO_APP
lui t0,%hi(nSieve)
lw a5,%lo(nSieve)(t0)
li a4,2
lui t5,%hi(primes)
lui t4,%hi(sieve)
addi t5,t5,%lo(primes)
addi t4,t4,%lo(sieve)
addiw t3,a5,1
sw a4,0(t5)
li a4,4
sw t3,%lo(nSieve)(t0)
li t2,0
sw a4,0(t4)
li a7,2
li a3,3
li a2,1
.L2:
mulw a5,a7,a7
ble a5,a3,.L3
addiw a7,a7,-1
ble t3,zero,.L23
mv a6,t5
mv a0,t4
slli t6,t3,2
sh2add t1,t3,t4
.L13:
lw a1,0(a6)
blt a7,a1,.L7
lw a5,0(a0)
ble a3,a5,.L9
.L8:
addw a5,a1,a5
bgt a3,a5,.L8
sw a5,0(a0)
.L9:
beq a3,a5,.L12
addi a0,a0,4
addi a6,a6,4
bne a0,t1,.L13
.L23:
beq t2,zero,.L6
sw t3,%lo(nSieve)(t0)
.L6:
#APP
# 19 "primes.c" 1
rdtime s0
# 0 "" 2
#NO_APP
lui a1,%hi(.LC1)
addi a1,a1,%lo(.LC1)
sub s0,s0,s1
li a0,2
call __printf_chk
fcvt.d.lu fa5,s0
lui a1,%hi(.LC2)
addi a1,a1,%lo(.LC2)
li a0,2
fmv.x.d a2,fa5
call __printf_chk
ld ra,24(sp)
.cfi_remember_state
.cfi_restore 1
li a0,0
ld s0,16(sp)
.cfi_restore 8
ld s1,8(sp)
.cfi_restore 9
addi sp,sp,32
.cfi_def_cfa_offset 0
jr ra
.L3:
.cfi_restore_state
addiw a7,a7,1
j .L2
.L7:
li a5,999
bgt t3,a5,.L11
mulw a5,a3,a3
add a4,t5,t6
add t6,t4,t6
sw a3,0(a4)
addiw t3,t3,1
li t2,1
sw a5,0(t6)
.L11:
addiw a2,a2,1
.L12:
addiw a3,a3,1
j .L2
.cfi_endproc
.LFE55:
.size main, .-main
.globl nSieve
.globl sieve
.globl primes
.bss
.align 3
.type sieve, @object
.size sieve, 4000
sieve:
.zero 4000
.type primes, @object
.size primes, 4000
primes:
.zero 4000
.section .sbss,"aw",@nobits
.align 2
.type nSieve, @object
.size nSieve, 4
nSieve:
.zero 4
.ident "GCC: (Bianbu 13.2.0-23ubuntu4bb3) 13.2.0"
.section .note.GNU-stack,"",@progbits
# lscpu
Architecture: riscv64
Byte Order: Little Endian
CPU(s): 8
On-line CPU(s) list: 0-7
Model name: Spacemit(R) X60
Thread(s) per core: 1
Core(s) per socket: 8
Socket(s): 1
CPU(s) scaling MHz: 100%
CPU max MHz: 1600.0000
CPU min MHz: 614.4000
Caches (sum of all):
L1d: 256 KiB (8 instances)
L1i: 256 KiB (8 instances)
L2: 1 MiB (2 instances)