Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
26 commits
Select commit Hold shift + click to select a range
9c3ba20
benchtool: bench-hash added
michalkucharczyk Jul 24, 2026
bc2efaa
bench-hash: initial version added
michalkucharczyk Jul 24, 2026
6d8903e
bench-hash: more algos added
michalkucharczyk Jul 24, 2026
9c97288
memcpy: poc
michalkucharczyk Jul 27, 2026
35ba9de
memcpy: feature added
michalkucharczyk Jul 27, 2026
5ab207c
added: +unaligned-scalar-mem
michalkucharczyk Jul 27, 2026
139c787
Merge remote-tracking branch 'origin/master' into mku-bench-hash
michalkucharczyk Jul 28, 2026
528becf
crypto benchmarks added
michalkucharczyk Jul 29, 2026
12db32d
bench-ed25519: use builtins-mem
michalkucharczyk Jul 29, 2026
b7b52c3
process-results
michalkucharczyk Jul 29, 2026
e41d0ca
naming cleanup
michalkucharczyk Jul 29, 2026
0ee9d98
experiments: blake2 asm+compact implementation added
michalkucharczyk Jul 29, 2026
59e3293
experiments: blake2 ...
michalkucharczyk Jul 29, 2026
1d575fb
benchtool: size and set-size added
michalkucharczyk Jul 30, 2026
05721e9
bench-hash removed
michalkucharczyk Jul 30, 2026
70a44be
bench-hash: crates
michalkucharczyk Jul 30, 2026
03538a9
blake2 asm tests added
michalkucharczyk Jul 30, 2026
ad6fde5
hash-bench-common added
michalkucharczyk Jul 31, 2026
6557a13
build-benchmarks.sh - new benchmarks + fixed toolchain
michalkucharczyk Jul 31, 2026
cd230d3
build-hash-native: target-cpu=native versions of host hash algos
michalkucharczyk Jul 31, 2026
a725c0b
Cargo.toml + lock
michalkucharczyk Jul 31, 2026
c16a7c8
process-results utils
michalkucharczyk Jul 31, 2026
1326240
tools/hash-asm-tests/Cargo.lock
michalkucharczyk Jul 31, 2026
1ef7253
some local utils
michalkucharczyk Jul 31, 2026
5f91214
investigating crypto host
michalkucharczyk Aug 3, 2026
3ff5439
recent nightly used
michalkucharczyk Aug 3, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
577 changes: 566 additions & 11 deletions guest-programs/Cargo.lock

Large diffs are not rendered by default.

16 changes: 16 additions & 0 deletions guest-programs/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,21 @@ members = [
"bench-minimal",
"bench-memset",
"bench-ed25519",
"bench-ed25519-zebra",
"bench-sr25519",
"bench-ecdsa-k256",
"bench-ecdsa-libsecp",
"bench-recover-k256",
"bench-recover-libsecp",
"bench-blake2-128",
"bench-blake2-256",
"bench-blake2-256-asm",
"bench-keccak-256",
"bench-keccak-512",
"bench-sha2-256",
"bench-twox-64",
"bench-twox-128",
"bench-twox-256",
"bench-pinky",
"bench-prime-sieve",

Expand All @@ -27,6 +42,7 @@ members = [

# Tests
"test-blob",
"test-blake2-asm",
]

[workspace.dependencies]
Expand Down
29 changes: 29 additions & 0 deletions guest-programs/bench-blake2-128/Cargo.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
[package]
name = "bench-blake2-128"
version = "0.1.0"
edition = "2021"
publish = false

[lib]
name = "bench_blake2_128"
path = "src/main.rs"
crate-type = ["cdylib"]

[[bin]]
name = "bench-blake2-128"
path = "src/main.rs"

[dependencies]
blake2b_simd = { version = "1", default-features = false }
picoalloc = { workspace = true }

[target.'cfg(target_env = "polkavm")'.dependencies]
polkavm-derive = { path = "../../crates/polkavm-derive" }

[features]
# Link compiler_builtins' word-wise memcpy/memset (what real-world guests get).
default = ["builtins-mem"]
builtins-mem = []

[lints]
workspace = true
12 changes: 12 additions & 0 deletions guest-programs/bench-blake2-128/src/main.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
#![no_std]
#![no_main]

include!("../../bench-common.rs");
include!("../../hash-bench-common.rs");

// 16-byte blake2b, as `sp_io::hashing::blake2_128`.
fn hash_once(input: &[u8], out: &mut [u8; 64]) -> usize {
let hash = blake2b_simd::Params::new().hash_length(16).hash(input);
out[..16].copy_from_slice(hash.as_bytes());
16
}
28 changes: 28 additions & 0 deletions guest-programs/bench-blake2-256-asm/Cargo.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
[package]
name = "bench-blake2-256-asm"
version = "0.1.0"
edition = "2021"
publish = false

[lib]
name = "bench_blake2_256_asm"
path = "src/main.rs"
crate-type = ["cdylib"]

[[bin]]
name = "bench-blake2-256-asm"
path = "src/main.rs"

[dependencies]
picoalloc = { workspace = true }

[target.'cfg(target_env = "polkavm")'.dependencies]
polkavm-derive = { path = "../../crates/polkavm-derive" }

[features]
# Link compiler_builtins' word-wise memcpy/memset (what real-world guests get).
default = ["builtins-mem"]
builtins-mem = []

[lints]
workspace = true
156 changes: 156 additions & 0 deletions guest-programs/bench-blake2-256-asm/src/asm_blake2b.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,156 @@
// blake2b-256 with a hand-scheduled RISC-V assembly compression function
// (`blake2b_compress.S`, generated).
//
// Rationale: the recompiled output of the portable Rust implementation is
// issue-bound — LLVM's register allocation for the 13-register rv64e target
// spills far more than necessary and its 3-operand code forces the
// recompiler to emit register-copy `mov`s. The assembly version pins the
// state's a/b rows (v0..v7) in registers, streams the c/d rows through a
// 64-byte stack frame, uses strictly two-operand form (no recompiler movs),
// and bakes the message schedule into constant load offsets.
//
// On non-riscv64 targets (host builds, riscv32) this falls back to a plain
// Rust implementation (a rounds loop; merged from the former
// `compact_blake2b` module) so the export exists everywhere and outputs
// stay comparable.

const IV: [u64; 8] = [
0x6a09e667f3bcc908,
0xbb67ae8584caa73b,
0x3c6ef372fe94f82b,
0xa54ff53a5f1d36f1,
0x510e527fade682d1,
0x9b05688c2b3e6c1f,
0x1f83d9abfb41bd6b,
0x5be0cd19137e2179,
];

#[cfg(target_arch = "riscv64")]
core::arch::global_asm!(include_str!("blake2b_compress.S"));

#[cfg(target_arch = "riscv64")]
extern "C" {
/// Processes `nblocks` consecutive 128-byte blocks; the byte counter for
/// block k is `t_base + 128*k` (wrapping). `last != 0` finalizes every
/// block in the batch, so callers pass `nblocks = 1` for the last block.
fn blake2b_compress_many(h: *mut u64, data: *const u8, nblocks: u64, t_base: u64, iv: *const u64, last: u64);
}

#[cfg(target_arch = "riscv64")]
pub fn blake2b_256(input: &[u8]) -> [u8; 32] {
let mut h = IV;
h[0] ^= 0x0101_0000 ^ 32;

// All full blocks except a full last block (the last block is always
// finalized separately below).
let full = input.len().saturating_sub(1) / 128;
if full > 0 {
unsafe { blake2b_compress_many(h.as_mut_ptr(), input.as_ptr(), full as u64, 0, IV.as_ptr(), 0) };
}

let rem = &input[full * 128..];
let mut block = [0u8; 128];
block[..rem.len()].copy_from_slice(rem);
let t_base = (input.len() as u64).wrapping_sub(128);
unsafe { blake2b_compress_many(h.as_mut_ptr(), block.as_ptr(), 1, t_base, IV.as_ptr(), 1) };

let mut out = [0u8; 32];
for i in 0..4 {
out[i * 8..][..8].copy_from_slice(&h[i].to_le_bytes());
}
out
}

#[cfg(not(target_arch = "riscv64"))]
const SIGMA: [[u8; 16]; 10] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15],
[14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3],
[11, 8, 12, 0, 5, 2, 15, 13, 10, 14, 3, 6, 7, 1, 9, 4],
[7, 9, 3, 1, 13, 12, 11, 14, 2, 6, 5, 10, 4, 0, 15, 8],
[9, 0, 5, 7, 2, 4, 10, 15, 14, 1, 11, 12, 6, 8, 3, 13],
[2, 12, 6, 10, 0, 11, 8, 3, 4, 13, 7, 5, 15, 14, 1, 9],
[12, 5, 1, 15, 14, 13, 4, 10, 0, 7, 6, 3, 9, 2, 8, 11],
[13, 11, 7, 14, 12, 1, 3, 9, 5, 0, 15, 4, 8, 6, 2, 10],
[6, 15, 14, 9, 11, 3, 0, 8, 12, 2, 13, 7, 1, 4, 10, 5],
[10, 2, 8, 4, 7, 6, 1, 5, 15, 11, 9, 14, 3, 12, 13, 0],
];

#[cfg(not(target_arch = "riscv64"))]
#[inline(always)]
fn g(v: &mut [u64; 16], a: usize, b: usize, c: usize, d: usize, x: u64, y: u64) {
v[a] = v[a].wrapping_add(v[b]).wrapping_add(x);
v[d] = (v[d] ^ v[a]).rotate_right(32);
v[c] = v[c].wrapping_add(v[d]);
v[b] = (v[b] ^ v[c]).rotate_right(24);
v[a] = v[a].wrapping_add(v[b]).wrapping_add(y);
v[d] = (v[d] ^ v[a]).rotate_right(16);
v[c] = v[c].wrapping_add(v[d]);
v[b] = (v[b] ^ v[c]).rotate_right(63);
}

#[cfg(not(target_arch = "riscv64"))]
fn compress(h: &mut [u64; 8], block: &[u8; 128], t: u128, last: bool) {
let mut m = [0u64; 16];
for (i, chunk) in block.chunks_exact(8).enumerate() {
m[i] = u64::from_le_bytes(chunk.try_into().unwrap());
}

let mut v = [0u64; 16];
v[..8].copy_from_slice(h);
v[8..].copy_from_slice(&IV);
v[12] ^= t as u64;
v[13] ^= (t >> 64) as u64;
if last {
v[14] = !v[14];
}

let mut s_idx = 0;
for _ in 0..12 {
let s = &SIGMA[s_idx];
s_idx += 1;
if s_idx == 10 {
s_idx = 0;
}
g(&mut v, 0, 4, 8, 12, m[s[0] as usize & 15], m[s[1] as usize & 15]);
g(&mut v, 1, 5, 9, 13, m[s[2] as usize & 15], m[s[3] as usize & 15]);
g(&mut v, 2, 6, 10, 14, m[s[4] as usize & 15], m[s[5] as usize & 15]);
g(&mut v, 3, 7, 11, 15, m[s[6] as usize & 15], m[s[7] as usize & 15]);
g(&mut v, 0, 5, 10, 15, m[s[8] as usize & 15], m[s[9] as usize & 15]);
g(&mut v, 1, 6, 11, 12, m[s[10] as usize & 15], m[s[11] as usize & 15]);
g(&mut v, 2, 7, 8, 13, m[s[12] as usize & 15], m[s[13] as usize & 15]);
g(&mut v, 3, 4, 9, 14, m[s[14] as usize & 15], m[s[15] as usize & 15]);
}

for i in 0..8 {
h[i] ^= v[i] ^ v[i + 8];
}
}

/// Plain-Rust fallback (a rounds loop; merged from the former
/// `compact_blake2b` module) for targets without the assembly.
#[cfg(not(target_arch = "riscv64"))]
pub fn blake2b_256(input: &[u8]) -> [u8; 32] {
let mut h = IV;
h[0] ^= 0x0101_0000 ^ 32; // digest_length = 32, key = 0, fanout = 1, depth = 1

let mut t: u128 = 0;
let mut offset = 0;
while input.len() - offset > 128 {
let block: &[u8; 128] = input[offset..offset + 128].try_into().unwrap();
t += 128;
compress(&mut h, block, t, false);
offset += 128;
}

let rem = &input[offset..];
let mut block = [0u8; 128];
block[..rem.len()].copy_from_slice(rem);
t += rem.len() as u128;
compress(&mut h, &block, t, true);

let mut out = [0u8; 32];
for i in 0..4 {
out[i * 8..][..8].copy_from_slice(&h[i].to_le_bytes());
}
out
}
Loading
Loading