diff --git a/zerocopy/benches/ref_from_bytes_dynamic_padding.x86-64 b/zerocopy/benches/ref_from_bytes_dynamic_padding.x86-64 index e844a4608f..c42f016ca9 100644 --- a/zerocopy/benches/ref_from_bytes_dynamic_padding.x86-64 +++ b/zerocopy/benches/ref_from_bytes_dynamic_padding.x86-64 @@ -1,22 +1,22 @@ bench_ref_from_bytes_dynamic_padding: test dil, 3 - jne .LBB5_3 + jne .LBB5_1 movabs rax, 9223372036854775804 and rax, rsi cmp rax, 9 - jb .LBB5_3 - add rax, -9 + setb al + test sil, 3 + setne cl + or cl, al + jne .LBB5_1 + add rsi, -9 movabs rcx, -6148914691236517205 + mov rax, rsi mul rcx shr rdx - lea rax, [rdx + 2*rdx] - or rax, 3 - add rax, 9 - cmp rsi, rax - je .LBB5_4 -.LBB5_3: + mov rax, rdi + ret +.LBB5_1: xor edi, edi - mov rdx, rsi -.LBB5_4: mov rax, rdi ret diff --git a/zerocopy/benches/ref_from_bytes_dynamic_padding.x86-64.mca b/zerocopy/benches/ref_from_bytes_dynamic_padding.x86-64.mca index 423ed38ba2..b020ff13e1 100644 --- a/zerocopy/benches/ref_from_bytes_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/ref_from_bytes_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1900 -Total Cycles: 645 -Total uOps: 2000 +Instructions: 2000 +Total Cycles: 673 +Total uOps: 2100 Dispatch Width: 4 -uOps Per Cycle: 3.10 -IPC: 2.95 -Block RThroughput: 5.0 +uOps Per Cycle: 3.12 +IPC: 2.97 +Block RThroughput: 5.3 Instruction Info: @@ -19,22 +19,23 @@ Instruction Info: [1] [2] [3] [4] [5] [6] Instructions: 1 1 0.33 test dil, 3 - 1 1 1.00 jne .LBB5_3 + 1 1 1.00 jne .LBB5_1 1 1 0.33 movabs rax, 9223372036854775804 1 1 0.33 and rax, rsi 1 1 0.33 cmp rax, 9 - 1 1 1.00 jb .LBB5_3 - 1 1 0.33 add rax, -9 + 1 1 0.50 setb al + 1 1 0.33 test sil, 3 + 1 1 0.50 setne cl + 1 1 0.33 or cl, al + 1 1 1.00 jne .LBB5_1 + 1 1 0.33 add rsi, -9 1 1 0.33 movabs rcx, -6148914691236517205 + 1 1 0.33 mov rax, rsi 2 4 1.00 mul rcx 1 1 0.50 shr rdx - 1 1 0.50 lea rax, [rdx + 2*rdx] - 1 1 0.33 or rax, 3 - 1 1 0.33 add rax, 9 - 1 1 0.33 cmp rsi, rax - 1 1 1.00 je .LBB5_4 + 1 1 0.33 mov rax, rdi + 1 1 1.00 U ret 1 0 0.25 xor edi, edi - 1 1 0.33 mov rdx, rsi 1 1 0.33 mov rax, rdi 1 1 1.00 U ret @@ -52,26 +53,27 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.32 6.33 - 6.35 - - + - - 6.66 6.66 - 6.68 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.64 0.35 - 0.01 - - test dil, 3 - - - - - - 1.00 - - jne .LBB5_3 - - - 0.34 0.65 - 0.01 - - movabs rax, 9223372036854775804 - - - 0.35 0.65 - - - - and rax, rsi - - - 0.33 0.34 - 0.33 - - cmp rax, 9 - - - - - - 1.00 - - jb .LBB5_3 - - - 0.35 - - 0.65 - - add rax, -9 - - - 0.97 0.01 - 0.02 - - movabs rcx, -6148914691236517205 + - - 0.66 0.33 - 0.01 - - test dil, 3 + - - - - - 1.00 - - jne .LBB5_1 + - - 0.33 0.67 - - - - movabs rax, 9223372036854775804 + - - 1.00 - - - - - and rax, rsi + - - 0.67 - - 0.33 - - cmp rax, 9 + - - 0.34 - - 0.66 - - setb al + - - - 1.00 - - - - test sil, 3 + - - 0.33 - - 0.67 - - setne cl + - - 1.00 - - - - - or cl, al + - - - - - 1.00 - - jne .LBB5_1 + - - - 1.00 - - - - add rsi, -9 + - - - 0.99 - 0.01 - - movabs rcx, -6148914691236517205 + - - - 0.34 - 0.66 - - mov rax, rsi - - 1.00 1.00 - - - - mul rcx - - - 0.99 - - 0.01 - - shr rdx - - - 0.33 0.67 - - - - lea rax, [rdx + 2*rdx] - - - 0.34 0.66 - - - - or rax, 3 - - - 0.33 0.66 - 0.01 - - add rax, 9 - - - 0.01 0.99 - - - - cmp rsi, rax - - - - - - 1.00 - - je .LBB5_4 + - - 1.00 - - - - - shr rdx + - - - 0.66 - 0.34 - - mov rax, rdi + - - - - - 1.00 - - ret - - - - - - - - xor edi, edi - - - 0.32 0.01 - 0.67 - - mov rdx, rsi - - - 0.02 0.34 - 0.64 - - mov rax, rdi + - - 0.33 0.67 - - - - mov rax, rdi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/ref_from_bytes_dynamic_size.x86-64 b/zerocopy/benches/ref_from_bytes_dynamic_size.x86-64 index cc905b76c0..87bf910afc 100644 --- a/zerocopy/benches/ref_from_bytes_dynamic_size.x86-64 +++ b/zerocopy/benches/ref_from_bytes_dynamic_size.x86-64 @@ -1,20 +1,16 @@ bench_ref_from_bytes_dynamic_size: + mov rax, rdi + test al, 1 + jne .LBB5_1 mov rdx, rsi cmp rsi, 4 - setb al - or al, dil - test al, 1 - je .LBB5_2 - xor eax, eax + setb cl + or sil, cl + test sil, 1 + jne .LBB5_1 + add rdx, -4 + shr rdx ret -.LBB5_2: - lea rcx, [rdx - 4] - mov rsi, rcx - and rsi, -2 - add rsi, 4 - shr rcx +.LBB5_1: xor eax, eax - cmp rdx, rsi - cmove rdx, rcx - cmove rax, rdi ret diff --git a/zerocopy/benches/ref_from_bytes_dynamic_size.x86-64.mca b/zerocopy/benches/ref_from_bytes_dynamic_size.x86-64.mca index 68aea583e4..41c86a6337 100644 --- a/zerocopy/benches/ref_from_bytes_dynamic_size.x86-64.mca +++ b/zerocopy/benches/ref_from_bytes_dynamic_size.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1800 -Total Cycles: 704 -Total uOps: 2000 +Instructions: 1400 +Total Cycles: 438 +Total uOps: 1400 Dispatch Width: 4 -uOps Per Cycle: 2.84 -IPC: 2.56 -Block RThroughput: 5.0 +uOps Per Cycle: 3.20 +IPC: 3.20 +Block RThroughput: 4.0 Instruction Info: @@ -18,23 +18,19 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: + 1 1 0.33 mov rax, rdi + 1 1 0.33 test al, 1 + 1 1 1.00 jne .LBB5_1 1 1 0.33 mov rdx, rsi 1 1 0.33 cmp rsi, 4 - 1 1 0.50 setb al - 1 1 0.33 or al, dil - 1 1 0.33 test al, 1 - 1 1 1.00 je .LBB5_2 - 1 0 0.25 xor eax, eax + 1 1 0.50 setb cl + 1 1 0.33 or sil, cl + 1 1 0.33 test sil, 1 + 1 1 1.00 jne .LBB5_1 + 1 1 0.33 add rdx, -4 + 1 1 0.50 shr rdx 1 1 1.00 U ret - 1 1 0.50 lea rcx, [rdx - 4] - 1 1 0.33 mov rsi, rcx - 1 1 0.33 and rsi, -2 - 1 1 0.33 add rsi, 4 - 1 1 0.50 shr rcx 1 0 0.25 xor eax, eax - 1 1 0.33 cmp rdx, rsi - 2 2 0.67 cmove rdx, rcx - 2 2 0.67 cmove rax, rdi 1 1 1.00 U ret @@ -51,25 +47,21 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 5.97 5.98 - 6.05 - - + - - 4.32 4.33 - 4.35 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.97 0.01 - 0.02 - - mov rdx, rsi - - - 0.01 0.02 - 0.97 - - cmp rsi, 4 - - - 0.03 - - 0.97 - - setb al - - - 0.01 0.02 - 0.97 - - or al, dil - - - - 0.98 - 0.02 - - test al, 1 - - - - - - 1.00 - - je .LBB5_2 - - - - - - - - - xor eax, eax + - - 0.33 0.66 - 0.01 - - mov rax, rdi + - - 0.67 0.33 - - - - test al, 1 + - - - - - 1.00 - - jne .LBB5_1 + - - 0.66 0.34 - - - - mov rdx, rsi + - - 0.33 0.66 - 0.01 - - cmp rsi, 4 + - - 1.00 - - - - - setb cl + - - - 1.00 - - - - or sil, cl + - - - 1.00 - - - - test sil, 1 + - - - - - 1.00 - - jne .LBB5_1 + - - 0.66 0.34 - - - - add rdx, -4 + - - 0.67 - - 0.33 - - shr rdx - - - - - 1.00 - - ret - - - 0.98 0.02 - - - - lea rcx, [rdx - 4] - - - 0.01 0.99 - - - - mov rsi, rcx - - - - 0.98 - 0.02 - - and rsi, -2 - - - 0.98 0.01 - 0.01 - - add rsi, 4 - - - 0.99 - - 0.01 - - shr rcx - - - - - - - - xor eax, eax - - - 0.02 0.97 - 0.01 - - cmp rdx, rsi - - - 0.99 0.99 - 0.02 - - cmove rdx, rcx - - - 0.98 0.99 - 0.03 - - cmove rax, rdi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64 b/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64 index a58592a245..f380fee6c2 100644 --- a/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64 +++ b/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64 @@ -1,16 +1,16 @@ bench_ref_from_prefix_dynamic_padding: - xor edx, edx - mov eax, 0 test dil, 3 - je .LBB5_1 - ret -.LBB5_1: + jne .LBB5_2 movabs rax, 9223372036854775804 and rsi, rax - cmp rsi, 9 - jae .LBB5_3 - mov edx, 1 - xor eax, eax + cmp rsi, 8 + ja .LBB5_3 +.LBB5_2: + xor edx, edx + test dil, 3 + sete dl + xor edi, edi + mov rax, rdi ret .LBB5_3: add rsi, -9 diff --git a/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64.mca b/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64.mca index 62ea4babaf..499c58d94b 100644 --- a/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64.mca @@ -1,6 +1,6 @@ Iterations: 100 Instructions: 1900 -Total Cycles: 608 +Total Cycles: 607 Total uOps: 2000 Dispatch Width: 4 @@ -18,17 +18,17 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 0 0.25 xor edx, edx - 1 1 0.33 mov eax, 0 1 1 0.33 test dil, 3 - 1 1 1.00 je .LBB5_1 - 1 1 1.00 U ret + 1 1 1.00 jne .LBB5_2 1 1 0.33 movabs rax, 9223372036854775804 1 1 0.33 and rsi, rax - 1 1 0.33 cmp rsi, 9 - 1 1 1.00 jae .LBB5_3 - 1 1 0.33 mov edx, 1 - 1 0 0.25 xor eax, eax + 1 1 0.33 cmp rsi, 8 + 1 1 1.00 ja .LBB5_3 + 1 0 0.25 xor edx, edx + 1 1 0.33 test dil, 3 + 1 1 0.50 sete dl + 1 0 0.25 xor edi, edi + 1 1 0.33 mov rax, rdi 1 1 1.00 U ret 1 1 0.33 add rsi, -9 1 1 0.33 movabs rcx, -6148914691236517205 @@ -52,26 +52,26 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.00 6.00 - 6.00 - - + - - 6.00 5.99 - 6.01 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: + - - 0.01 0.98 - 0.01 - - test dil, 3 + - - - - - 1.00 - - jne .LBB5_2 + - - 0.98 0.01 - 0.01 - - movabs rax, 9223372036854775804 + - - 0.01 0.99 - - - - and rsi, rax + - - 0.01 0.99 - - - - cmp rsi, 8 + - - - - - 1.00 - - ja .LBB5_3 - - - - - - - - xor edx, edx - - - 0.01 0.98 - 0.01 - - mov eax, 0 - - - 0.98 0.01 - 0.01 - - test dil, 3 - - - - - - 1.00 - - je .LBB5_1 - - - - - - 1.00 - - ret - - - 0.01 0.99 - - - - movabs rax, 9223372036854775804 - - - - 1.00 - - - - and rsi, rax - - - - 1.00 - - - - cmp rsi, 9 - - - - - - 1.00 - - jae .LBB5_3 - - - 1.00 - - - - - mov edx, 1 - - - - - - - - - xor eax, eax + - - 0.99 0.01 - - - - test dil, 3 + - - 0.99 - - 0.01 - - sete dl + - - - - - - - - xor edi, edi + - - - 0.01 - 0.99 - - mov rax, rdi - - - - - 1.00 - - ret - - - 0.02 0.02 - 0.96 - - add rsi, -9 - - - 0.99 0.01 - - - - movabs rcx, -6148914691236517205 - - - 0.01 0.99 - - - - mov rax, rsi + - - 0.01 0.99 - - - - add rsi, -9 + - - - 1.00 - - - - movabs rcx, -6148914691236517205 + - - 1.00 - - - - - mov rax, rsi - - 1.00 1.00 - - - - mul rcx - - 1.00 - - - - - shr rdx - - - 0.98 - - 0.02 - - mov rax, rdi + - - - 0.01 - 0.99 - - mov rax, rdi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64 b/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64 index fe6332c910..3f2e874678 100644 --- a/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64 +++ b/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64 @@ -1,17 +1,10 @@ bench_ref_from_prefix_dynamic_size: - xor edx, edx - mov eax, 0 - test dil, 1 - jne .LBB5_4 - cmp rsi, 4 - jae .LBB5_3 - mov edx, 1 - xor eax, eax - ret -.LBB5_3: - add rsi, -4 - shr rsi mov rdx, rsi - mov rax, rdi -.LBB5_4: + xor eax, eax + sub rdx, 4 + mov rcx, rdi + cmovb rcx, rax + shr rdx + test dil, 1 + cmove rax, rcx ret diff --git a/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64.mca b/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64.mca index 3900a59461..7ee51e0b8a 100644 --- a/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64.mca +++ b/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1400 -Total Cycles: 405 -Total uOps: 1400 +Instructions: 900 +Total Cycles: 339 +Total uOps: 1100 Dispatch Width: 4 -uOps Per Cycle: 3.46 -IPC: 3.46 -Block RThroughput: 4.0 +uOps Per Cycle: 3.24 +IPC: 2.65 +Block RThroughput: 2.8 Instruction Info: @@ -18,19 +18,14 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 0 0.25 xor edx, edx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test dil, 1 - 1 1 1.00 jne .LBB5_4 - 1 1 0.33 cmp rsi, 4 - 1 1 1.00 jae .LBB5_3 - 1 1 0.33 mov edx, 1 - 1 0 0.25 xor eax, eax - 1 1 1.00 U ret - 1 1 0.33 add rsi, -4 - 1 1 0.50 shr rsi 1 1 0.33 mov rdx, rsi - 1 1 0.33 mov rax, rdi + 1 0 0.25 xor eax, eax + 1 1 0.33 sub rdx, 4 + 1 1 0.33 mov rcx, rdi + 2 2 0.67 cmovb rcx, rax + 1 1 0.50 shr rdx + 1 1 0.33 test dil, 1 + 2 2 0.67 cmove rax, rcx 1 1 1.00 U ret @@ -47,21 +42,16 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 3.99 3.99 - 4.02 - - + - - 3.33 3.33 - 3.34 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - - - - - - - xor edx, edx - - - 0.01 0.98 - 0.01 - - mov eax, 0 - - - 0.98 0.02 - - - - test dil, 1 - - - - - - 1.00 - - jne .LBB5_4 - - - 0.02 0.98 - - - - cmp rsi, 4 - - - - - - 1.00 - - jae .LBB5_3 - - - 0.98 0.01 - 0.01 - - mov edx, 1 + - - 0.33 0.66 - 0.01 - - mov rdx, rsi - - - - - - - - xor eax, eax - - - - - - 1.00 - - ret - - - 0.01 0.99 - - - - add rsi, -4 - - - 1.00 - - - - - shr rsi - - - - 1.00 - - - - mov rdx, rsi - - - 0.99 0.01 - - - - mov rax, rdi + - - 0.01 - - 0.99 - - sub rdx, 4 + - - 0.66 0.34 - - - - mov rcx, rdi + - - 1.00 1.00 - - - - cmovb rcx, rax + - - 0.33 - - 0.67 - - shr rdx + - - - 0.33 - 0.67 - - test dil, 1 + - - 1.00 1.00 - - - - cmove rax, rcx - - - - - 1.00 - - ret diff --git a/zerocopy/benches/ref_from_suffix_dynamic_padding.x86-64 b/zerocopy/benches/ref_from_suffix_dynamic_padding.x86-64 index 3e05f6023f..754261ca5a 100644 --- a/zerocopy/benches/ref_from_suffix_dynamic_padding.x86-64 +++ b/zerocopy/benches/ref_from_suffix_dynamic_padding.x86-64 @@ -7,17 +7,15 @@ bench_ref_from_suffix_dynamic_padding: cmp rax, 9 jae .LBB5_3 .LBB5_1: - xor eax, eax + xor edi, edi + mov rax, rdi ret .LBB5_3: add rax, -9 movabs rcx, -6148914691236517205 mul rcx shr rdx - lea rax, [rdx + 2*rdx] - sub rsi, rax - or rax, -4 - add rsi, rdi - add rax, rsi - add rax, -8 + and esi, 3 + add rdi, rsi + mov rax, rdi ret diff --git a/zerocopy/benches/ref_from_suffix_dynamic_padding.x86-64.mca b/zerocopy/benches/ref_from_suffix_dynamic_padding.x86-64.mca index 73599d5b6a..c2f825d836 100644 --- a/zerocopy/benches/ref_from_suffix_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/ref_from_suffix_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2000 -Total Cycles: 682 -Total uOps: 2100 +Instructions: 1800 +Total Cycles: 606 +Total uOps: 1900 Dispatch Width: 4 -uOps Per Cycle: 3.08 -IPC: 2.93 -Block RThroughput: 5.3 +uOps Per Cycle: 3.14 +IPC: 2.97 +Block RThroughput: 4.8 Instruction Info: @@ -25,18 +25,16 @@ Instruction Info: 1 1 0.33 and rax, rsi 1 1 0.33 cmp rax, 9 1 1 1.00 jae .LBB5_3 - 1 0 0.25 xor eax, eax + 1 0 0.25 xor edi, edi + 1 1 0.33 mov rax, rdi 1 1 1.00 U ret 1 1 0.33 add rax, -9 1 1 0.33 movabs rcx, -6148914691236517205 2 4 1.00 mul rcx 1 1 0.50 shr rdx - 1 1 0.50 lea rax, [rdx + 2*rdx] - 1 1 0.33 sub rsi, rax - 1 1 0.33 or rax, -4 - 1 1 0.33 add rsi, rdi - 1 1 0.33 add rax, rsi - 1 1 0.33 add rax, -8 + 1 1 0.33 and esi, 3 + 1 1 0.33 add rdi, rsi + 1 1 0.33 mov rax, rdi 1 1 1.00 U ret @@ -53,27 +51,25 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.65 6.67 - 6.68 - - + - - 6.00 6.00 - 6.00 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.90 0.10 - - - - lea eax, [rsi + rdi] - - - 0.93 - - 0.07 - - test al, 3 + - - 0.99 0.01 - - - - lea eax, [rsi + rdi] + - - 0.01 0.99 - - - - test al, 3 - - - - - 1.00 - - jne .LBB5_1 - - - 0.51 0.47 - 0.02 - - movabs rax, 9223372036854775804 - - - - - - 1.00 - - and rax, rsi - - - - 0.09 - 0.91 - - cmp rax, 9 + - - 0.99 - - 0.01 - - movabs rax, 9223372036854775804 + - - 0.99 - - 0.01 - - and rax, rsi + - - - 0.96 - 0.04 - - cmp rax, 9 - - - - - 1.00 - - jae .LBB5_3 - - - - - - - - - xor eax, eax + - - - - - - - - xor edi, edi + - - 0.01 - - 0.99 - - mov rax, rdi - - - - - 1.00 - - ret - - - 0.43 0.47 - 0.10 - - add rax, -9 - - - 0.42 0.39 - 0.19 - - movabs rcx, -6148914691236517205 + - - 0.95 0.05 - - - - add rax, -9 + - - 0.05 0.95 - - - - movabs rcx, -6148914691236517205 - - 1.00 1.00 - - - - mul rcx - - - 0.69 - - 0.31 - - shr rdx - - - 0.54 0.46 - - - - lea rax, [rdx + 2*rdx] - - - 0.07 0.91 - 0.02 - - sub rsi, rax - - - 0.91 0.05 - 0.04 - - or rax, -4 - - - 0.08 0.90 - 0.02 - - add rsi, rdi - - - 0.09 0.91 - - - - add rax, rsi - - - 0.08 0.92 - - - - add rax, -8 + - - 1.00 - - - - - shr rdx + - - - 1.00 - - - - and esi, 3 + - - 0.01 0.04 - 0.95 - - add rdi, rsi + - - - 1.00 - - - - mov rax, rdi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64 b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64 index c3d10b5fc6..7d11849dd5 100644 --- a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64 +++ b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64 @@ -1,27 +1,21 @@ bench_ref_from_suffix_with_elems_dynamic_padding: - movabs rax, 3074457345618258598 - cmp rdx, rax - ja .LBB5_1 - lea r8d, [rsi + rdi] - xor ecx, ecx - mov eax, 0 - test r8b, 3 - je .LBB5_3 - mov rdx, rcx - ret + movabs rcx, 3074457345618258598 + cmp rdx, rcx + ja .LBB5_3 + mov rax, rdi + lea rcx, [rdx + 2*rdx] + or rcx, 3 + add rcx, 9 + add edi, esi + test dil, 3 + setne dil + sub rsi, rcx + setb cl + or cl, dil + je .LBB5_4 .LBB5_3: - lea rax, [rdx + 2*rdx] - or rax, 3 - add rax, 9 - sub rsi, rax - jae .LBB5_4 -.LBB5_1: xor eax, eax - mov edx, 1 ret .LBB5_4: - add rdi, rsi - mov rcx, rdx - mov rax, rdi - mov rdx, rcx + add rax, rsi ret diff --git a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64.mca b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64.mca index 92e6280bb4..d7850d93bc 100644 --- a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2300 -Total Cycles: 706 -Total uOps: 2300 +Instructions: 1800 +Total Cycles: 572 +Total uOps: 1800 Dispatch Width: 4 -uOps Per Cycle: 3.26 -IPC: 3.26 -Block RThroughput: 6.0 +uOps Per Cycle: 3.15 +IPC: 3.15 +Block RThroughput: 4.5 Instruction Info: @@ -18,28 +18,23 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 1 0.33 movabs rax, 3074457345618258598 - 1 1 0.33 cmp rdx, rax - 1 1 1.00 ja .LBB5_1 - 1 1 0.50 lea r8d, [rsi + rdi] - 1 0 0.25 xor ecx, ecx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test r8b, 3 - 1 1 1.00 je .LBB5_3 - 1 1 0.33 mov rdx, rcx - 1 1 1.00 U ret - 1 1 0.50 lea rax, [rdx + 2*rdx] - 1 1 0.33 or rax, 3 - 1 1 0.33 add rax, 9 - 1 1 0.33 sub rsi, rax - 1 1 1.00 jae .LBB5_4 + 1 1 0.33 movabs rcx, 3074457345618258598 + 1 1 0.33 cmp rdx, rcx + 1 1 1.00 ja .LBB5_3 + 1 1 0.33 mov rax, rdi + 1 1 0.50 lea rcx, [rdx + 2*rdx] + 1 1 0.33 or rcx, 3 + 1 1 0.33 add rcx, 9 + 1 1 0.33 add edi, esi + 1 1 0.33 test dil, 3 + 1 1 0.50 setne dil + 1 1 0.33 sub rsi, rcx + 1 1 0.50 setb cl + 1 1 0.33 or cl, dil + 1 1 1.00 je .LBB5_4 1 0 0.25 xor eax, eax - 1 1 0.33 mov edx, 1 1 1 1.00 U ret - 1 1 0.33 add rdi, rsi - 1 1 0.33 mov rcx, rdx - 1 1 0.33 mov rax, rdi - 1 1 0.33 mov rdx, rcx + 1 1 0.33 add rax, rsi 1 1 1.00 U ret @@ -56,30 +51,25 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.99 7.00 - 7.01 - - + - - 5.66 5.66 - 5.68 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - - 0.99 - 0.01 - - movabs rax, 3074457345618258598 - - - 0.01 0.50 - 0.49 - - cmp rdx, rax - - - - - - 1.00 - - ja .LBB5_1 - - - - 1.00 - - - - lea r8d, [rsi + rdi] - - - - - - - - - xor ecx, ecx - - - 0.50 0.49 - 0.01 - - mov eax, 0 - - - 0.49 0.51 - - - - test r8b, 3 - - - - - - 1.00 - - je .LBB5_3 - - - 0.51 0.49 - - - - mov rdx, rcx - - - - - - 1.00 - - ret - - - 0.50 0.50 - - - - lea rax, [rdx + 2*rdx] - - - 1.00 - - - - - or rax, 3 - - - 1.00 - - - - - add rax, 9 - - - 0.99 0.01 - - - - sub rsi, rax - - - - - - 1.00 - - jae .LBB5_4 + - - 0.67 - - 0.33 - - movabs rcx, 3074457345618258598 + - - 0.34 - - 0.66 - - cmp rdx, rcx + - - - - - 1.00 - - ja .LBB5_3 + - - 0.33 0.01 - 0.66 - - mov rax, rdi + - - 0.33 0.67 - - - - lea rcx, [rdx + 2*rdx] + - - 0.01 0.99 - - - - or rcx, 3 + - - 0.01 0.99 - - - - add rcx, 9 + - - 0.99 - - 0.01 - - add edi, esi + - - 0.99 0.01 - - - - test dil, 3 + - - 0.99 - - 0.01 - - setne dil + - - - 1.00 - - - - sub rsi, rcx + - - 1.00 - - - - - setb cl + - - - 0.99 - 0.01 - - or cl, dil + - - - - - 1.00 - - je .LBB5_4 - - - - - - - - xor eax, eax - - - - 1.00 - - - - mov edx, 1 - - - - - 1.00 - - ret - - - 1.00 - - - - - add rdi, rsi - - - - 1.00 - - - - mov rcx, rdx - - - 0.99 0.01 - - - - mov rax, rdi - - - - 0.50 - 0.50 - - mov rdx, rcx + - - - 1.00 - - - - add rax, rsi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64 b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64 index bdca571924..7ee0278dbf 100644 --- a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64 +++ b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64 @@ -1,23 +1,18 @@ bench_ref_from_suffix_with_elems_dynamic_size: - movabs rax, 4611686018427387901 - cmp rdx, rax - ja .LBB5_1 - lea r8d, [rsi + rdi] - xor ecx, ecx - mov eax, 0 - test r8b, 1 - jne .LBB5_5 - lea rax, [2*rdx + 4] - sub rsi, rax - jae .LBB5_4 -.LBB5_1: + movabs rcx, 4611686018427387901 + cmp rdx, rcx + ja .LBB5_3 + mov rax, rdi + lea rcx, [2*rdx + 4] + add edi, esi + sub rsi, rcx + setb cl + or cl, dil + test cl, 1 + je .LBB5_4 +.LBB5_3: xor eax, eax - mov edx, 1 ret .LBB5_4: - add rdi, rsi - mov rcx, rdx - mov rax, rdi -.LBB5_5: - mov rdx, rcx + add rax, rsi ret diff --git a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64.mca b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64.mca index 6d9de0b3eb..882316de3d 100644 --- a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64.mca +++ b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1900 -Total Cycles: 571 -Total uOps: 1900 +Instructions: 1500 +Total Cycles: 472 +Total uOps: 1500 Dispatch Width: 4 -uOps Per Cycle: 3.33 -IPC: 3.33 -Block RThroughput: 5.0 +uOps Per Cycle: 3.18 +IPC: 3.18 +Block RThroughput: 4.0 Instruction Info: @@ -18,24 +18,20 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 1 0.33 movabs rax, 4611686018427387901 - 1 1 0.33 cmp rdx, rax - 1 1 1.00 ja .LBB5_1 - 1 1 0.50 lea r8d, [rsi + rdi] - 1 0 0.25 xor ecx, ecx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test r8b, 1 - 1 1 1.00 jne .LBB5_5 - 1 1 0.50 lea rax, [2*rdx + 4] - 1 1 0.33 sub rsi, rax - 1 1 1.00 jae .LBB5_4 + 1 1 0.33 movabs rcx, 4611686018427387901 + 1 1 0.33 cmp rdx, rcx + 1 1 1.00 ja .LBB5_3 + 1 1 0.33 mov rax, rdi + 1 1 0.50 lea rcx, [2*rdx + 4] + 1 1 0.33 add edi, esi + 1 1 0.33 sub rsi, rcx + 1 1 0.50 setb cl + 1 1 0.33 or cl, dil + 1 1 0.33 test cl, 1 + 1 1 1.00 je .LBB5_4 1 0 0.25 xor eax, eax - 1 1 0.33 mov edx, 1 1 1 1.00 U ret - 1 1 0.33 add rdi, rsi - 1 1 0.33 mov rcx, rdx - 1 1 0.33 mov rax, rdi - 1 1 0.33 mov rdx, rcx + 1 1 0.33 add rax, rsi 1 1 1.00 U ret @@ -52,26 +48,22 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 5.66 5.66 - 5.68 - - + - - 4.66 4.66 - 4.68 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.66 0.33 - 0.01 - - movabs rax, 4611686018427387901 - - - 0.01 0.99 - - - - cmp rdx, rax - - - - - - 1.00 - - ja .LBB5_1 - - - 0.99 0.01 - - - - lea r8d, [rsi + rdi] - - - - - - - - - xor ecx, ecx - - - 0.33 0.33 - 0.34 - - mov eax, 0 - - - 0.33 0.34 - 0.33 - - test r8b, 1 - - - - - - 1.00 - - jne .LBB5_5 - - - 0.34 0.66 - - - - lea rax, [2*rdx + 4] - - - - 1.00 - - - - sub rsi, rax - - - - - - 1.00 - - jae .LBB5_4 + - - 0.99 - - 0.01 - - movabs rcx, 4611686018427387901 + - - 1.00 - - - - - cmp rdx, rcx + - - - - - 1.00 - - ja .LBB5_3 + - - 0.63 0.02 - 0.35 - - mov rax, rdi + - - 0.35 0.65 - - - - lea rcx, [2*rdx + 4] + - - 0.65 0.03 - 0.32 - - add edi, esi + - - 0.03 0.97 - - - - sub rsi, rcx + - - 1.00 - - - - - setb cl + - - - 1.00 - - - - or cl, dil + - - 0.01 0.99 - - - - test cl, 1 + - - - - - 1.00 - - je .LBB5_4 - - - - - - - - xor eax, eax - - - 1.00 - - - - - mov edx, 1 - - - - - 1.00 - - ret - - - - 1.00 - - - - add rdi, rsi - - - 1.00 - - - - - mov rcx, rdx - - - 0.32 0.68 - - - - mov rax, rdi - - - 0.68 0.32 - - - - mov rdx, rcx + - - - 1.00 - - - - add rax, rsi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_read_from_bytes.x86-64 b/zerocopy/benches/try_read_from_bytes.x86-64 index 08088a08fd..2135facc26 100644 --- a/zerocopy/benches/try_read_from_bytes.x86-64 +++ b/zerocopy/benches/try_read_from_bytes.x86-64 @@ -1,23 +1,11 @@ bench_try_read_from_bytes_static_size: - mov ax, -16191 + mov eax, 49345 cmp rsi, 6 - jne .LBB5_1 - mov ecx, dword ptr [rdi] - movzx edx, cx - cmp edx, 49344 - jne .LBB5_4 - movzx eax, word ptr [rdi + 4] - shl rax, 32 - or rcx, rax - shr rcx, 16 - mov ax, -16192 -.LBB5_4: - shl rcx, 16 - movzx eax, ax - or rax, rcx - ret -.LBB5_1: - shl rcx, 16 - movzx eax, ax - or rax, rcx + jne .LBB5_3 + cmp word ptr [rdi], -16192 + jne .LBB5_3 + mov eax, dword ptr [rdi + 2] + shl rax, 16 + or rax, 49344 +.LBB5_3: ret diff --git a/zerocopy/benches/try_read_from_bytes.x86-64.mca b/zerocopy/benches/try_read_from_bytes.x86-64.mca index 385e6a4802..f5568fef5a 100644 --- a/zerocopy/benches/try_read_from_bytes.x86-64.mca +++ b/zerocopy/benches/try_read_from_bytes.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2000 -Total Cycles: 608 -Total uOps: 2000 +Instructions: 900 +Total Cycles: 305 +Total uOps: 1000 Dispatch Width: 4 -uOps Per Cycle: 3.29 -IPC: 3.29 -Block RThroughput: 5.0 +uOps Per Cycle: 3.28 +IPC: 2.95 +Block RThroughput: 3.0 Instruction Info: @@ -18,25 +18,14 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 1 0.33 mov ax, -16191 + 1 1 0.33 mov eax, 49345 1 1 0.33 cmp rsi, 6 - 1 1 1.00 jne .LBB5_1 - 1 5 0.50 * mov ecx, dword ptr [rdi] - 1 1 0.33 movzx edx, cx - 1 1 0.33 cmp edx, 49344 - 1 1 1.00 jne .LBB5_4 - 1 5 0.50 * movzx eax, word ptr [rdi + 4] - 1 1 0.50 shl rax, 32 - 1 1 0.33 or rcx, rax - 1 1 0.50 shr rcx, 16 - 1 1 0.33 mov ax, -16192 - 1 1 0.50 shl rcx, 16 - 1 1 0.33 movzx eax, ax - 1 1 0.33 or rax, rcx - 1 1 1.00 U ret - 1 1 0.50 shl rcx, 16 - 1 1 0.33 movzx eax, ax - 1 1 0.33 or rax, rcx + 1 1 1.00 jne .LBB5_3 + 2 6 0.50 * cmp word ptr [rdi], -16192 + 1 1 1.00 jne .LBB5_3 + 1 5 0.50 * mov eax, dword ptr [rdi + 2] + 1 1 0.50 shl rax, 16 + 1 1 0.33 or rax, 49344 1 1 1.00 U ret @@ -53,27 +42,16 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 5.99 5.99 - 6.02 1.00 1.00 + - - 2.50 2.49 - 3.01 1.00 1.00 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - - 0.99 - 0.01 - - mov ax, -16191 - - - - 0.01 - 0.99 - - cmp rsi, 6 - - - - - - 1.00 - - jne .LBB5_1 - - - - - - - - 1.00 mov ecx, dword ptr [rdi] - - - 0.98 - - 0.02 - - movzx edx, cx - - - 0.99 0.01 - - - - cmp edx, 49344 - - - - - - 1.00 - - jne .LBB5_4 - - - - - - - 1.00 - movzx eax, word ptr [rdi + 4] - - - 0.01 - - 0.99 - - shl rax, 32 - - - 0.02 0.98 - - - - or rcx, rax - - - 1.00 - - - - - shr rcx, 16 - - - 0.99 0.01 - - - - mov ax, -16192 - - - 1.00 - - - - - shl rcx, 16 - - - - 1.00 - - - - movzx eax, ax - - - - 1.00 - - - - or rax, rcx - - - - - - 1.00 - - ret - - - 1.00 - - - - - shl rcx, 16 - - - - 1.00 - - - - movzx eax, ax - - - - 0.99 - 0.01 - - or rax, rcx + - - 0.50 0.49 - 0.01 - - mov eax, 49345 + - - 0.49 0.51 - - - - cmp rsi, 6 + - - - - - 1.00 - - jne .LBB5_3 + - - 0.47 0.53 - - 0.43 0.57 cmp word ptr [rdi], -16192 + - - - - - 1.00 - - jne .LBB5_3 + - - - - - - 0.57 0.43 mov eax, dword ptr [rdi + 2] + - - 1.00 - - - - - shl rax, 16 + - - 0.04 0.96 - - - - or rax, 49344 - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_read_from_prefix.x86-64 b/zerocopy/benches/try_read_from_prefix.x86-64 index d3e1edc3ea..0f820efff5 100644 --- a/zerocopy/benches/try_read_from_prefix.x86-64 +++ b/zerocopy/benches/try_read_from_prefix.x86-64 @@ -1,16 +1,11 @@ bench_try_read_from_prefix_static_size: mov eax, 49345 cmp rsi, 6 - jb .LBB5_2 - mov eax, dword ptr [rdi] - movzx ecx, word ptr [rdi + 4] - shl rcx, 32 - or rcx, rax - movzx eax, cx - and rcx, -65536 - or rcx, 49344 - cmp eax, 49344 - mov eax, 49345 - cmove rax, rcx -.LBB5_2: + jb .LBB5_3 + cmp word ptr [rdi], -16192 + jne .LBB5_3 + mov eax, dword ptr [rdi + 2] + shl rax, 16 + or rax, 49344 +.LBB5_3: ret diff --git a/zerocopy/benches/try_read_from_prefix.x86-64.mca b/zerocopy/benches/try_read_from_prefix.x86-64.mca index 40401d89e8..e0ba4c8320 100644 --- a/zerocopy/benches/try_read_from_prefix.x86-64.mca +++ b/zerocopy/benches/try_read_from_prefix.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1400 -Total Cycles: 442 -Total uOps: 1500 +Instructions: 900 +Total Cycles: 305 +Total uOps: 1000 Dispatch Width: 4 -uOps Per Cycle: 3.39 -IPC: 3.17 -Block RThroughput: 3.8 +uOps Per Cycle: 3.28 +IPC: 2.95 +Block RThroughput: 3.0 Instruction Info: @@ -20,17 +20,12 @@ Instruction Info: [1] [2] [3] [4] [5] [6] Instructions: 1 1 0.33 mov eax, 49345 1 1 0.33 cmp rsi, 6 - 1 1 1.00 jb .LBB5_2 - 1 5 0.50 * mov eax, dword ptr [rdi] - 1 5 0.50 * movzx ecx, word ptr [rdi + 4] - 1 1 0.50 shl rcx, 32 - 1 1 0.33 or rcx, rax - 1 1 0.33 movzx eax, cx - 1 1 0.33 and rcx, -65536 - 1 1 0.33 or rcx, 49344 - 1 1 0.33 cmp eax, 49344 - 1 1 0.33 mov eax, 49345 - 2 2 0.67 cmove rax, rcx + 1 1 1.00 jb .LBB5_3 + 2 6 0.50 * cmp word ptr [rdi], -16192 + 1 1 1.00 jne .LBB5_3 + 1 5 0.50 * mov eax, dword ptr [rdi + 2] + 1 1 0.50 shl rax, 16 + 1 1 0.33 or rax, 49344 1 1 1.00 U ret @@ -47,21 +42,16 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 4.33 4.33 - 4.34 1.00 1.00 + - - 2.50 2.49 - 3.01 1.00 1.00 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.65 0.01 - 0.34 - - mov eax, 49345 - - - 0.01 0.33 - 0.66 - - cmp rsi, 6 - - - - - - 1.00 - - jb .LBB5_2 - - - - - - - - 1.00 mov eax, dword ptr [rdi] - - - - - - - 1.00 - movzx ecx, word ptr [rdi + 4] - - - 0.65 - - 0.35 - - shl rcx, 32 - - - - 0.67 - 0.33 - - or rcx, rax - - - 0.01 0.99 - - - - movzx eax, cx - - - 0.99 0.01 - - - - and rcx, -65536 - - - 0.01 0.99 - - - - or rcx, 49344 - - - 0.99 0.01 - - - - cmp eax, 49344 - - - 0.02 0.33 - 0.65 - - mov eax, 49345 - - - 1.00 0.99 - 0.01 - - cmove rax, rcx + - - 0.50 0.49 - 0.01 - - mov eax, 49345 + - - 0.49 0.51 - - - - cmp rsi, 6 + - - - - - 1.00 - - jb .LBB5_3 + - - 0.47 0.53 - - 0.43 0.57 cmp word ptr [rdi], -16192 + - - - - - 1.00 - - jne .LBB5_3 + - - - - - - 0.57 0.43 mov eax, dword ptr [rdi + 2] + - - 1.00 - - - - - shl rax, 16 + - - 0.04 0.96 - - - - or rax, 49344 - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_read_from_suffix.x86-64 b/zerocopy/benches/try_read_from_suffix.x86-64 index 095e326f04..a91c074e18 100644 --- a/zerocopy/benches/try_read_from_suffix.x86-64 +++ b/zerocopy/benches/try_read_from_suffix.x86-64 @@ -1,18 +1,11 @@ bench_try_read_from_suffix_static_size: mov eax, 49345 cmp rsi, 6 - jb .LBB5_2 - mov eax, dword ptr [rdi + rsi - 6] - movzx ecx, word ptr [rdi + rsi - 2] - shl rcx, 32 - or rcx, rax - movzx edx, cx - xor eax, eax - cmp edx, 49344 - cmovne rcx, rsi - sete al - and rcx, -65536 - xor rax, 49345 - or rax, rcx -.LBB5_2: + jb .LBB5_3 + cmp word ptr [rdi + rsi - 6], -16192 + jne .LBB5_3 + mov eax, dword ptr [rdi + rsi - 4] + shl rax, 16 + or rax, 49344 +.LBB5_3: ret diff --git a/zerocopy/benches/try_read_from_suffix.x86-64.mca b/zerocopy/benches/try_read_from_suffix.x86-64.mca index d3eaadbb8a..e42f6b93ef 100644 --- a/zerocopy/benches/try_read_from_suffix.x86-64.mca +++ b/zerocopy/benches/try_read_from_suffix.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1600 -Total Cycles: 478 -Total uOps: 1700 +Instructions: 900 +Total Cycles: 305 +Total uOps: 1000 Dispatch Width: 4 -uOps Per Cycle: 3.56 -IPC: 3.35 -Block RThroughput: 4.3 +uOps Per Cycle: 3.28 +IPC: 2.95 +Block RThroughput: 3.0 Instruction Info: @@ -20,19 +20,12 @@ Instruction Info: [1] [2] [3] [4] [5] [6] Instructions: 1 1 0.33 mov eax, 49345 1 1 0.33 cmp rsi, 6 - 1 1 1.00 jb .LBB5_2 - 1 5 0.50 * mov eax, dword ptr [rdi + rsi - 6] - 1 5 0.50 * movzx ecx, word ptr [rdi + rsi - 2] - 1 1 0.50 shl rcx, 32 - 1 1 0.33 or rcx, rax - 1 1 0.33 movzx edx, cx - 1 0 0.25 xor eax, eax - 1 1 0.33 cmp edx, 49344 - 2 2 0.67 cmovne rcx, rsi - 1 1 0.50 sete al - 1 1 0.33 and rcx, -65536 - 1 1 0.33 xor rax, 49345 - 1 1 0.33 or rax, rcx + 1 1 1.00 jb .LBB5_3 + 2 6 0.50 * cmp word ptr [rdi + rsi - 6], -16192 + 1 1 1.00 jne .LBB5_3 + 1 5 0.50 * mov eax, dword ptr [rdi + rsi - 4] + 1 1 0.50 shl rax, 16 + 1 1 0.33 or rax, 49344 1 1 1.00 U ret @@ -49,23 +42,16 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 4.66 4.66 - 4.68 1.00 1.00 + - - 2.50 2.49 - 3.01 1.00 1.00 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.32 0.01 - 0.67 - - mov eax, 49345 - - - 0.62 0.02 - 0.36 - - cmp rsi, 6 - - - - - - 1.00 - - jb .LBB5_2 - - - - - - - - 1.00 mov eax, dword ptr [rdi + rsi - 6] - - - - - - - 1.00 - movzx ecx, word ptr [rdi + rsi - 2] - - - 0.37 - - 0.63 - - shl rcx, 32 - - - 0.99 0.01 - - - - or rcx, rax - - - 1.00 - - - - - movzx edx, cx - - - - - - - - - xor eax, eax - - - 0.35 0.64 - 0.01 - - cmp edx, 49344 - - - 1.00 1.00 - - - - cmovne rcx, rsi - - - - - - 1.00 - - sete al - - - 0.01 0.99 - - - - and rcx, -65536 - - - - 1.00 - - - - xor rax, 49345 - - - - 0.99 - 0.01 - - or rax, rcx + - - 0.50 0.49 - 0.01 - - mov eax, 49345 + - - 0.49 0.51 - - - - cmp rsi, 6 + - - - - - 1.00 - - jb .LBB5_3 + - - 0.47 0.53 - - 0.43 0.57 cmp word ptr [rdi + rsi - 6], -16192 + - - - - - 1.00 - - jne .LBB5_3 + - - - - - - 0.57 0.43 mov eax, dword ptr [rdi + rsi - 4] + - - 1.00 - - - - - shl rax, 16 + - - 0.04 0.96 - - - - or rax, 49344 - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_bytes_dynamic_padding.x86-64 b/zerocopy/benches/try_ref_from_bytes_dynamic_padding.x86-64 index 217c5fc617..0782220491 100644 --- a/zerocopy/benches/try_ref_from_bytes_dynamic_padding.x86-64 +++ b/zerocopy/benches/try_ref_from_bytes_dynamic_padding.x86-64 @@ -1,24 +1,24 @@ bench_try_ref_from_bytes_dynamic_padding: - test dil, 3 - jne .LBB5_4 movabs rax, 9223372036854775804 and rax, rsi cmp rax, 9 - jb .LBB5_4 - add rax, -9 + setae al + mov ecx, esi + or ecx, edi + test cl, 3 + sete cl + and cl, al + cmp cl, 1 + jne .LBB5_1 + add rsi, -9 movabs rcx, -6148914691236517205 + mov rax, rsi mul rcx shr rdx - lea rax, [rdx + 2*rdx] - or rax, 3 - add rax, 9 - cmp rsi, rax - jne .LBB5_4 + xor eax, eax cmp word ptr [rdi], -16192 - je .LBB5_5 -.LBB5_4: - xor edi, edi - mov rdx, rsi -.LBB5_5: - mov rax, rdi + cmove rax, rdi + ret +.LBB5_1: + xor eax, eax ret diff --git a/zerocopy/benches/try_ref_from_bytes_dynamic_padding.x86-64.mca b/zerocopy/benches/try_ref_from_bytes_dynamic_padding.x86-64.mca index 95b993c7e0..ed0586281f 100644 --- a/zerocopy/benches/try_ref_from_bytes_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/try_ref_from_bytes_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2100 -Total Cycles: 709 -Total uOps: 2300 +Instructions: 2200 +Total Cycles: 742 +Total uOps: 2500 Dispatch Width: 4 -uOps Per Cycle: 3.24 +uOps Per Cycle: 3.37 IPC: 2.96 -Block RThroughput: 5.8 +Block RThroughput: 6.3 Instruction Info: @@ -18,26 +18,27 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 1 0.33 test dil, 3 - 1 1 1.00 jne .LBB5_4 1 1 0.33 movabs rax, 9223372036854775804 1 1 0.33 and rax, rsi 1 1 0.33 cmp rax, 9 - 1 1 1.00 jb .LBB5_4 - 1 1 0.33 add rax, -9 + 1 1 0.50 setae al + 1 1 0.33 mov ecx, esi + 1 1 0.33 or ecx, edi + 1 1 0.33 test cl, 3 + 1 1 0.50 sete cl + 1 1 0.33 and cl, al + 1 1 0.33 cmp cl, 1 + 1 1 1.00 jne .LBB5_1 + 1 1 0.33 add rsi, -9 1 1 0.33 movabs rcx, -6148914691236517205 + 1 1 0.33 mov rax, rsi 2 4 1.00 mul rcx 1 1 0.50 shr rdx - 1 1 0.50 lea rax, [rdx + 2*rdx] - 1 1 0.33 or rax, 3 - 1 1 0.33 add rax, 9 - 1 1 0.33 cmp rsi, rax - 1 1 1.00 jne .LBB5_4 + 1 0 0.25 xor eax, eax 2 6 0.50 * cmp word ptr [rdi], -16192 - 1 1 1.00 je .LBB5_5 - 1 0 0.25 xor edi, edi - 1 1 0.33 mov rdx, rsi - 1 1 0.33 mov rax, rdi + 2 2 0.67 cmove rax, rdi + 1 1 1.00 U ret + 1 0 0.25 xor eax, eax 1 1 1.00 U ret @@ -54,28 +55,29 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.98 6.99 - 7.03 0.50 0.50 + - - 7.32 7.33 - 7.35 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.48 0.51 - 0.01 - - test dil, 3 - - - - - - 1.00 - - jne .LBB5_4 - - - 0.51 0.49 - - - - movabs rax, 9223372036854775804 - - - 0.01 0.99 - - - - and rax, rsi - - - 0.51 0.49 - - - - cmp rax, 9 - - - - - - 1.00 - - jb .LBB5_4 - - - 0.98 - - 0.02 - - add rax, -9 - - - 0.98 0.02 - - - - movabs rcx, -6148914691236517205 + - - - 0.99 - 0.01 - - movabs rax, 9223372036854775804 + - - 0.02 0.98 - - - - and rax, rsi + - - 0.67 0.32 - 0.01 - - cmp rax, 9 + - - 0.99 - - 0.01 - - setae al + - - 0.01 0.34 - 0.65 - - mov ecx, esi + - - 0.32 0.02 - 0.66 - - or ecx, edi + - - 0.32 0.67 - 0.01 - - test cl, 3 + - - 0.33 - - 0.67 - - sete cl + - - 0.99 - - 0.01 - - and cl, al + - - 0.98 0.01 - 0.01 - - cmp cl, 1 + - - - - - 1.00 - - jne .LBB5_1 + - - 0.01 0.99 - - - - add rsi, -9 + - - 0.01 0.35 - 0.64 - - movabs rcx, -6148914691236517205 + - - - 0.98 - 0.02 - - mov rax, rsi - - 1.00 1.00 - - - - mul rcx - - - 0.99 - - 0.01 - - shr rdx - - - - 1.00 - - - - lea rax, [rdx + 2*rdx] - - - - 0.51 - 0.49 - - or rax, 3 - - - 0.01 0.49 - 0.50 - - add rax, 9 - - - - 0.02 - 0.98 - - cmp rsi, rax - - - - - - 1.00 - - jne .LBB5_4 - - - 0.51 0.49 - - 0.50 0.50 cmp word ptr [rdi], -16192 - - - - - - 1.00 - - je .LBB5_5 - - - - - - - - - xor edi, edi - - - 0.50 0.50 - - - - mov rdx, rsi - - - 0.50 0.48 - 0.02 - - mov rax, rdi + - - 1.00 - - - - - shr rdx + - - - - - - - - xor eax, eax + - - 0.01 0.34 - 0.65 0.50 0.50 cmp word ptr [rdi], -16192 + - - 0.66 0.34 - 1.00 - - cmove rax, rdi + - - - - - 1.00 - - ret + - - - - - - - - xor eax, eax - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_bytes_dynamic_size.x86-64 b/zerocopy/benches/try_ref_from_bytes_dynamic_size.x86-64 index cf67afd31c..ab0336e767 100644 --- a/zerocopy/benches/try_ref_from_bytes_dynamic_size.x86-64 +++ b/zerocopy/benches/try_ref_from_bytes_dynamic_size.x86-64 @@ -1,22 +1,21 @@ bench_try_ref_from_bytes_dynamic_size: - mov rdx, rsi - mov rax, rdi cmp rsi, 4 - setb cl - or cl, al + setae al + mov ecx, esi + or ecx, edi test cl, 1 - jne .LBB5_4 - lea rcx, [rdx - 4] - mov rsi, rcx - and rsi, -2 - add rsi, 4 - cmp rdx, rsi - jne .LBB5_4 - cmp word ptr [rax], -16192 - jne .LBB5_4 - shr rcx - mov rdx, rcx + sete cl + and cl, al + cmp cl, 1 + jne .LBB5_1 + lea rdx, [rsi - 4] + shr rdx + movzx ecx, word ptr [rdi] + xor eax, eax + cmp ecx, 49344 + cmovne rdx, rsi + cmove rax, rdi ret -.LBB5_4: +.LBB5_1: xor eax, eax ret diff --git a/zerocopy/benches/try_ref_from_bytes_dynamic_size.x86-64.mca b/zerocopy/benches/try_ref_from_bytes_dynamic_size.x86-64.mca index ecd7a18f6d..c0baeab83a 100644 --- a/zerocopy/benches/try_ref_from_bytes_dynamic_size.x86-64.mca +++ b/zerocopy/benches/try_ref_from_bytes_dynamic_size.x86-64.mca @@ -1,10 +1,10 @@ Iterations: 100 -Instructions: 2000 -Total Cycles: 639 +Instructions: 1900 +Total Cycles: 607 Total uOps: 2100 Dispatch Width: 4 -uOps Per Cycle: 3.29 +uOps Per Cycle: 3.46 IPC: 3.13 Block RThroughput: 5.3 @@ -18,23 +18,22 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 1 0.33 mov rdx, rsi - 1 1 0.33 mov rax, rdi 1 1 0.33 cmp rsi, 4 - 1 1 0.50 setb cl - 1 1 0.33 or cl, al + 1 1 0.50 setae al + 1 1 0.33 mov ecx, esi + 1 1 0.33 or ecx, edi 1 1 0.33 test cl, 1 - 1 1 1.00 jne .LBB5_4 - 1 1 0.50 lea rcx, [rdx - 4] - 1 1 0.33 mov rsi, rcx - 1 1 0.33 and rsi, -2 - 1 1 0.33 add rsi, 4 - 1 1 0.33 cmp rdx, rsi - 1 1 1.00 jne .LBB5_4 - 2 6 0.50 * cmp word ptr [rax], -16192 - 1 1 1.00 jne .LBB5_4 - 1 1 0.50 shr rcx - 1 1 0.33 mov rdx, rcx + 1 1 0.50 sete cl + 1 1 0.33 and cl, al + 1 1 0.33 cmp cl, 1 + 1 1 1.00 jne .LBB5_1 + 1 1 0.50 lea rdx, [rsi - 4] + 1 1 0.50 shr rdx + 1 5 0.50 * movzx ecx, word ptr [rdi] + 1 0 0.25 xor eax, eax + 1 1 0.33 cmp ecx, 49344 + 2 2 0.67 cmovne rdx, rsi + 2 2 0.67 cmove rax, rdi 1 1 1.00 U ret 1 0 0.25 xor eax, eax 1 1 1.00 U ret @@ -53,27 +52,26 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.32 6.32 - 6.36 0.50 0.50 + - - 5.99 5.99 - 6.02 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.33 0.66 - 0.01 - - mov rdx, rsi - - - 0.66 0.34 - - - - mov rax, rdi - - - 0.34 0.66 - - - - cmp rsi, 4 - - - 0.99 - - 0.01 - - setb cl - - - 0.01 0.99 - - - - or cl, al - - - - 1.00 - - - - test cl, 1 - - - - - - 1.00 - - jne .LBB5_4 - - - 0.66 0.34 - - - - lea rcx, [rdx - 4] - - - 0.33 0.66 - 0.01 - - mov rsi, rcx - - - 1.00 - - - - - and rsi, -2 - - - 0.66 0.34 - - - - add rsi, 4 - - - - 1.00 - - - - cmp rdx, rsi - - - - - - 1.00 - - jne .LBB5_4 - - - - - - 1.00 0.50 0.50 cmp word ptr [rax], -16192 - - - - - - 1.00 - - jne .LBB5_4 - - - 0.67 - - 0.33 - - shr rcx - - - 0.67 0.33 - - - - mov rdx, rcx + - - 0.03 0.96 - 0.01 - - cmp rsi, 4 + - - 1.00 - - - - - setae al + - - 0.96 0.04 - - - - mov ecx, esi + - - 0.03 0.96 - 0.01 - - or ecx, edi + - - 0.03 0.96 - 0.01 - - test cl, 1 + - - 0.05 - - 0.95 - - sete cl + - - 0.01 0.03 - 0.96 - - and cl, al + - - - 0.02 - 0.98 - - cmp cl, 1 + - - - - - 1.00 - - jne .LBB5_1 + - - 0.97 0.03 - - - - lea rdx, [rsi - 4] + - - 1.00 - - - - - shr rdx + - - - - - - 0.50 0.50 movzx ecx, word ptr [rdi] + - - - - - - - - xor eax, eax + - - - 0.99 - 0.01 - - cmp ecx, 49344 + - - 0.96 1.00 - 0.04 - - cmovne rdx, rsi + - - 0.95 1.00 - 0.05 - - cmove rax, rdi - - - - - 1.00 - - ret - - - - - - - - xor eax, eax - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64 b/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64 index d832cb7ecf..aa4ddaaca7 100644 --- a/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64 +++ b/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64 @@ -1,15 +1,14 @@ bench_try_ref_from_prefix_dynamic_padding: - xor edx, edx - mov eax, 0 test dil, 3 - je .LBB5_1 - ret -.LBB5_1: + jne .LBB5_2 movabs rax, 9223372036854775804 and rsi, rax - cmp rsi, 9 - jae .LBB5_3 - mov edx, 1 + cmp rsi, 8 + ja .LBB5_3 +.LBB5_2: + xor edx, edx + test dil, 3 + sete dl xor eax, eax ret .LBB5_3: diff --git a/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64.mca b/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64.mca index 482112a39b..f595d499cf 100644 --- a/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2600 -Total Cycles: 843 -Total uOps: 2900 +Instructions: 2500 +Total Cycles: 808 +Total uOps: 2800 Dispatch Width: 4 -uOps Per Cycle: 3.44 -IPC: 3.08 -Block RThroughput: 7.3 +uOps Per Cycle: 3.47 +IPC: 3.09 +Block RThroughput: 7.0 Instruction Info: @@ -18,16 +18,15 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 0 0.25 xor edx, edx - 1 1 0.33 mov eax, 0 1 1 0.33 test dil, 3 - 1 1 1.00 je .LBB5_1 - 1 1 1.00 U ret + 1 1 1.00 jne .LBB5_2 1 1 0.33 movabs rax, 9223372036854775804 1 1 0.33 and rsi, rax - 1 1 0.33 cmp rsi, 9 - 1 1 1.00 jae .LBB5_3 - 1 1 0.33 mov edx, 1 + 1 1 0.33 cmp rsi, 8 + 1 1 1.00 ja .LBB5_3 + 1 0 0.25 xor edx, edx + 1 1 0.33 test dil, 3 + 1 1 0.50 sete dl 1 0 0.25 xor eax, eax 1 1 1.00 U ret 1 1 0.33 add rsi, -9 @@ -59,33 +58,32 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 8.33 8.33 - 8.34 0.50 0.50 + - - 8.00 8.00 - 8.00 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: + - - 0.97 0.02 - 0.01 - - test dil, 3 + - - - - - 1.00 - - jne .LBB5_2 + - - 0.99 0.01 - - - - movabs rax, 9223372036854775804 + - - 0.02 0.98 - - - - and rsi, rax + - - 0.03 0.97 - - - - cmp rsi, 8 + - - - - - 1.00 - - ja .LBB5_3 - - - - - - - - xor edx, edx - - - 0.32 0.34 - 0.34 - - mov eax, 0 - - - 0.34 0.33 - 0.33 - - test dil, 3 - - - - - - 1.00 - - je .LBB5_1 - - - - - - 1.00 - - ret - - - 0.35 0.65 - - - - movabs rax, 9223372036854775804 - - - 0.96 0.03 - 0.01 - - and rsi, rax - - - 0.01 0.97 - 0.02 - - cmp rsi, 9 - - - - - - 1.00 - - jae .LBB5_3 - - - 0.67 0.01 - 0.32 - - mov edx, 1 + - - 0.98 0.02 - - - - test dil, 3 + - - 0.99 - - 0.01 - - sete dl - - - - - - - - xor eax, eax - - - - - 1.00 - - ret - - - 0.02 0.34 - 0.64 - - add rsi, -9 - - - 0.33 0.66 - 0.01 - - movabs rcx, -6148914691236517205 - - - 0.66 0.34 - - - - mov rax, rsi + - - 0.98 0.02 - - - - add rsi, -9 + - - 0.02 0.98 - - - - movabs rcx, -6148914691236517205 + - - 0.98 0.02 - - - - mov rax, rsi - - 1.00 1.00 - - - - mul rcx - - - 0.01 0.99 - - - - mov rax, rdx - - - 0.99 - - 0.01 - - shr rax + - - - 0.01 - 0.99 - - mov rax, rdx + - - 0.01 - - 0.99 - - shr rax - - - - - - 0.50 0.50 movzx ecx, word ptr [rdi] - - - 0.33 0.03 - 0.64 - - cmp cx, -16192 - - - 0.01 0.31 - 0.68 - - mov edx, 2 - - - 1.00 1.00 - - - - cmove rdx, rax + - - - 0.99 - 0.01 - - cmp cx, -16192 + - - 0.99 0.01 - - - - mov edx, 2 + - - 0.01 1.00 - 0.99 - - cmove rdx, rax - - - - - - - - xor eax, eax - - - 0.33 0.33 - 0.34 - - cmp ecx, 49344 - - - 1.00 1.00 - - - - cmove rax, rdi + - - 0.02 0.98 - - - - cmp ecx, 49344 + - - 0.01 0.99 - 1.00 - - cmove rax, rdi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64 b/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64 index be7f34b9f8..7cd0a48b21 100644 --- a/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64 +++ b/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64 @@ -1,14 +1,15 @@ bench_try_ref_from_prefix_dynamic_size: - xor edx, edx - mov eax, 0 - test dil, 1 - jne .LBB5_4 cmp rsi, 4 - jae .LBB5_3 - mov edx, 1 + setb al + or al, dil + test al, 1 + je .LBB5_2 + and edi, 1 + xor rdi, 1 xor eax, eax + mov rdx, rdi ret -.LBB5_3: +.LBB5_2: add rsi, -4 shr rsi movzx ecx, word ptr [rdi] @@ -18,5 +19,4 @@ bench_try_ref_from_prefix_dynamic_size: xor eax, eax cmp cx, -16192 cmove rax, rdi -.LBB5_4: ret diff --git a/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64.mca b/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64.mca index 11706defe1..43fdb855ee 100644 --- a/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64.mca +++ b/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1900 -Total Cycles: 573 -Total uOps: 2100 +Instructions: 2000 +Total Cycles: 643 +Total uOps: 2200 Dispatch Width: 4 -uOps Per Cycle: 3.66 -IPC: 3.32 -Block RThroughput: 5.3 +uOps Per Cycle: 3.42 +IPC: 3.11 +Block RThroughput: 5.5 Instruction Info: @@ -18,14 +18,15 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 0 0.25 xor edx, edx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test dil, 1 - 1 1 1.00 jne .LBB5_4 1 1 0.33 cmp rsi, 4 - 1 1 1.00 jae .LBB5_3 - 1 1 0.33 mov edx, 1 + 1 1 0.50 setb al + 1 1 0.33 or al, dil + 1 1 0.33 test al, 1 + 1 1 1.00 je .LBB5_2 + 1 1 0.33 and edi, 1 + 1 1 0.33 xor rdi, 1 1 0 0.25 xor eax, eax + 1 1 0.33 mov rdx, rdi 1 1 1.00 U ret 1 1 0.33 add rsi, -4 1 1 0.50 shr rsi @@ -52,26 +53,27 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 5.66 5.67 - 5.67 0.50 0.50 + - - 6.33 6.33 - 6.34 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - - - - - - - xor edx, edx - - - 0.30 0.37 - 0.33 - - mov eax, 0 - - - 0.35 0.32 - 0.33 - - test dil, 1 - - - - - - 1.00 - - jne .LBB5_4 - - - 0.32 0.33 - 0.35 - - cmp rsi, 4 - - - - - - 1.00 - - jae .LBB5_3 - - - 0.33 0.35 - 0.32 - - mov edx, 1 + - - 0.67 0.32 - 0.01 - - cmp rsi, 4 + - - 0.99 - - 0.01 - - setb al + - - 0.19 0.48 - 0.33 - - or al, dil + - - 0.50 0.16 - 0.34 - - test al, 1 + - - - - - 1.00 - - je .LBB5_2 + - - 0.50 0.50 - - - - and edi, 1 + - - 0.49 0.34 - 0.17 - - xor rdi, 1 - - - - - - - - xor eax, eax + - - 0.16 0.84 - - - - mov rdx, rdi - - - - - 1.00 - - ret - - - 0.34 0.64 - 0.02 - - add rsi, -4 - - - 1.00 - - - - - shr rsi + - - 0.68 0.32 - - - - add rsi, -4 + - - 0.85 - - 0.15 - - shr rsi - - - - - - 0.50 0.50 movzx ecx, word ptr [rdi] - - - 0.60 0.40 - - - - cmp ecx, 49344 - - - 0.05 0.95 - - - - mov edx, 2 - - - 1.00 1.00 - - - - cmove rdx, rsi + - - - 0.50 - 0.50 - - cmp ecx, 49344 + - - 0.63 0.37 - - - - mov edx, 2 + - - 0.17 1.00 - 0.83 - - cmove rdx, rsi - - - - - - - - xor eax, eax - - - 0.37 0.31 - 0.32 - - cmp cx, -16192 - - - 1.00 1.00 - - - - cmove rax, rdi + - - 0.50 0.50 - - - - cmp cx, -16192 + - - - 1.00 - 1.00 - - cmove rax, rdi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_suffix_dynamic_padding.x86-64 b/zerocopy/benches/try_ref_from_suffix_dynamic_padding.x86-64 index b3e9244428..e585ae7bee 100644 --- a/zerocopy/benches/try_ref_from_suffix_dynamic_padding.x86-64 +++ b/zerocopy/benches/try_ref_from_suffix_dynamic_padding.x86-64 @@ -14,13 +14,9 @@ bench_try_ref_from_suffix_dynamic_padding: movabs rcx, -6148914691236517205 mul rcx shr rdx - lea rcx, [rdx + 2*rdx] - sub rsi, rcx - or rcx, -4 - add rsi, rdi - lea rdi, [rcx + rsi] - add rdi, -8 + and esi, 3 + lea rcx, [rdi + rsi] xor eax, eax - cmp word ptr [rcx + rsi - 8], -16192 - cmove rax, rdi + cmp word ptr [rdi + rsi], -16192 + cmove rax, rcx ret diff --git a/zerocopy/benches/try_ref_from_suffix_dynamic_padding.x86-64.mca b/zerocopy/benches/try_ref_from_suffix_dynamic_padding.x86-64.mca index d56ae56d85..5754aa0fdb 100644 --- a/zerocopy/benches/try_ref_from_suffix_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/try_ref_from_suffix_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2300 -Total Cycles: 791 -Total uOps: 2600 +Instructions: 1900 +Total Cycles: 644 +Total uOps: 2200 Dispatch Width: 4 -uOps Per Cycle: 3.29 -IPC: 2.91 -Block RThroughput: 6.5 +uOps Per Cycle: 3.42 +IPC: 2.95 +Block RThroughput: 5.5 Instruction Info: @@ -31,15 +31,11 @@ Instruction Info: 1 1 0.33 movabs rcx, -6148914691236517205 2 4 1.00 mul rcx 1 1 0.50 shr rdx - 1 1 0.50 lea rcx, [rdx + 2*rdx] - 1 1 0.33 sub rsi, rcx - 1 1 0.33 or rcx, -4 - 1 1 0.33 add rsi, rdi - 1 1 0.50 lea rdi, [rcx + rsi] - 1 1 0.33 add rdi, -8 + 1 1 0.33 and esi, 3 + 1 1 0.50 lea rcx, [rdi + rsi] 1 0 0.25 xor eax, eax - 2 6 0.50 * cmp word ptr [rcx + rsi - 8], -16192 - 2 2 0.67 cmove rax, rdi + 2 6 0.50 * cmp word ptr [rdi + rsi], -16192 + 2 2 0.67 cmove rax, rcx 1 1 1.00 U ret @@ -56,30 +52,26 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 7.70 7.58 - 7.72 0.50 0.50 + - - 6.33 6.33 - 6.34 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.26 0.74 - - - - lea eax, [rsi + rdi] - - - 0.19 0.28 - 0.53 - - test al, 3 + - - 0.33 0.67 - - - - lea eax, [rsi + rdi] + - - 0.99 - - 0.01 - - test al, 3 - - - - - 1.00 - - jne .LBB5_1 - - - 0.93 0.06 - 0.01 - - movabs rax, 9223372036854775804 - - - 0.81 0.14 - 0.05 - - and rax, rsi - - - 0.55 0.43 - 0.02 - - cmp rax, 9 + - - 0.01 0.66 - 0.33 - - movabs rax, 9223372036854775804 + - - 0.32 0.34 - 0.34 - - and rax, rsi + - - 0.65 0.03 - 0.32 - - cmp rax, 9 - - - - - 1.00 - - jae .LBB5_3 - - - - - - - - xor eax, eax - - - - - 1.00 - - ret - - - 0.42 0.56 - 0.02 - - add rax, -9 - - - 0.67 0.30 - 0.03 - - movabs rcx, -6148914691236517205 + - - 0.02 0.98 - - - - add rax, -9 + - - 0.35 0.65 - - - - movabs rcx, -6148914691236517205 - - 1.00 1.00 - - - - mul rcx - - - 0.71 - - 0.29 - - shr rdx - - - 0.32 0.68 - - - - lea rcx, [rdx + 2*rdx] - - - 0.57 0.04 - 0.39 - - sub rsi, rcx - - - 0.28 0.67 - 0.05 - - or rcx, -4 - - - 0.29 0.29 - 0.42 - - add rsi, rdi - - - 0.02 0.98 - - - - lea rdi, [rcx + rsi] - - - 0.02 0.41 - 0.57 - - add rdi, -8 + - - 1.00 - - - - - shr rdx + - - 0.16 0.84 - - - - and esi, 3 + - - 0.17 0.83 - - - - lea rcx, [rdi + rsi] - - - - - - - - xor eax, eax - - - 0.57 0.01 - 0.42 0.50 0.50 cmp word ptr [rcx + rsi - 8], -16192 - - - 0.09 0.99 - 0.92 - - cmove rax, rdi + - - 0.34 0.32 - 0.34 0.50 0.50 cmp word ptr [rdi + rsi], -16192 + - - 0.99 0.01 - 1.00 - - cmove rax, rcx - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64 b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64 index c7530d8b68..ca483dee63 100644 --- a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64 +++ b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64 @@ -1,32 +1,23 @@ bench_try_ref_from_suffix_with_elems_dynamic_padding: movabs rax, 3074457345618258598 cmp rdx, rax - ja .LBB5_1 - lea r8d, [rsi + rdi] - xor ecx, ecx - mov eax, 0 - test r8b, 3 - je .LBB5_3 - mov rdx, rcx - ret -.LBB5_3: + ja .LBB5_3 lea rax, [rdx + 2*rdx] or rax, 3 add rax, 9 + lea ecx, [rsi + rdi] + test cl, 3 + setne cl sub rsi, rax - jae .LBB5_4 -.LBB5_1: + setb al + or al, cl + je .LBB5_4 +.LBB5_3: xor eax, eax - mov edx, 1 ret .LBB5_4: - lea r8, [rdi + rsi] - movzx esi, word ptr [rdi + rsi] - cmp si, -16192 - mov ecx, 2 - cmove rcx, rdx + lea rcx, [rdi + rsi] xor eax, eax - cmp esi, 49344 - cmove rax, r8 - mov rdx, rcx + cmp word ptr [rdi + rsi], -16192 + cmove rax, rcx ret diff --git a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64.mca b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64.mca index be736c00c2..c37a42c160 100644 --- a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2800 -Total Cycles: 878 -Total uOps: 3000 +Instructions: 2000 +Total Cycles: 643 +Total uOps: 2200 Dispatch Width: 4 uOps Per Cycle: 3.42 -IPC: 3.19 -Block RThroughput: 7.5 +IPC: 3.11 +Block RThroughput: 5.5 Instruction Info: @@ -20,31 +20,23 @@ Instruction Info: [1] [2] [3] [4] [5] [6] Instructions: 1 1 0.33 movabs rax, 3074457345618258598 1 1 0.33 cmp rdx, rax - 1 1 1.00 ja .LBB5_1 - 1 1 0.50 lea r8d, [rsi + rdi] - 1 0 0.25 xor ecx, ecx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test r8b, 3 - 1 1 1.00 je .LBB5_3 - 1 1 0.33 mov rdx, rcx - 1 1 1.00 U ret + 1 1 1.00 ja .LBB5_3 1 1 0.50 lea rax, [rdx + 2*rdx] 1 1 0.33 or rax, 3 1 1 0.33 add rax, 9 + 1 1 0.50 lea ecx, [rsi + rdi] + 1 1 0.33 test cl, 3 + 1 1 0.50 setne cl 1 1 0.33 sub rsi, rax - 1 1 1.00 jae .LBB5_4 + 1 1 0.50 setb al + 1 1 0.33 or al, cl + 1 1 1.00 je .LBB5_4 1 0 0.25 xor eax, eax - 1 1 0.33 mov edx, 1 1 1 1.00 U ret - 1 1 0.50 lea r8, [rdi + rsi] - 1 5 0.50 * movzx esi, word ptr [rdi + rsi] - 1 1 0.33 cmp si, -16192 - 1 1 0.33 mov ecx, 2 - 2 2 0.67 cmove rcx, rdx + 1 1 0.50 lea rcx, [rdi + rsi] 1 0 0.25 xor eax, eax - 1 1 0.33 cmp esi, 49344 - 2 2 0.67 cmove rax, r8 - 1 1 0.33 mov rdx, rcx + 2 6 0.50 * cmp word ptr [rdi + rsi], -16192 + 2 2 0.67 cmove rax, rcx 1 1 1.00 U ret @@ -61,35 +53,27 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 8.65 8.65 - 8.70 0.50 0.50 + - - 6.32 6.33 - 6.35 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.67 0.30 - 0.03 - - movabs rax, 3074457345618258598 - - - 0.01 0.99 - - - - cmp rdx, rax - - - - - - 1.00 - - ja .LBB5_1 - - - 0.99 0.01 - - - - lea r8d, [rsi + rdi] - - - - - - - - - xor ecx, ecx - - - 0.35 0.62 - 0.03 - - mov eax, 0 - - - 0.99 0.01 - - - - test r8b, 3 - - - - - - 1.00 - - je .LBB5_3 - - - 0.68 0.30 - 0.02 - - mov rdx, rcx - - - - - - 1.00 - - ret - - - 0.07 0.93 - - - - lea rax, [rdx + 2*rdx] - - - 0.06 0.35 - 0.59 - - or rax, 3 - - - 0.02 0.07 - 0.91 - - add rax, 9 - - - 0.01 0.04 - 0.95 - - sub rsi, rax - - - - - - 1.00 - - jae .LBB5_4 + - - - 0.99 - 0.01 - - movabs rax, 3074457345618258598 + - - 0.02 0.98 - - - - cmp rdx, rax + - - - - - 1.00 - - ja .LBB5_3 + - - 0.66 0.34 - - - - lea rax, [rdx + 2*rdx] + - - 0.99 - - 0.01 - - or rax, 3 + - - 0.99 0.01 - - - - add rax, 9 + - - 0.01 0.99 - - - - lea ecx, [rsi + rdi] + - - - 0.99 - 0.01 - - test cl, 3 + - - 0.02 - - 0.98 - - setne cl + - - 0.97 0.02 - 0.01 - - sub rsi, rax + - - 0.99 - - 0.01 - - setb al + - - 0.32 0.67 - 0.01 - - or al, cl + - - - - - 1.00 - - je .LBB5_4 - - - - - - - - xor eax, eax - - - 0.92 0.01 - 0.07 - - mov edx, 1 - - - - - 1.00 - - ret - - - - 1.00 - - - - lea r8, [rdi + rsi] - - - - - - - 0.50 0.50 movzx esi, word ptr [rdi + rsi] - - - 0.01 0.99 - - - - cmp si, -16192 - - - 0.88 0.04 - 0.08 - - mov ecx, 2 - - - 1.00 0.99 - 0.01 - - cmove rcx, rdx + - - - 1.00 - - - - lea rcx, [rdi + rsi] - - - - - - - - xor eax, eax - - - 0.99 0.01 - - - - cmp esi, 49344 - - - 1.00 1.00 - - - - cmove rax, r8 - - - - 0.99 - 0.01 - - mov rdx, rcx + - - 0.35 0.32 - 0.33 0.50 0.50 cmp word ptr [rdi + rsi], -16192 + - - 1.00 0.02 - 0.98 - - cmove rax, rcx - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64 b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64 index 952eb12de8..748eb1166e 100644 --- a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64 +++ b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64 @@ -1,28 +1,20 @@ bench_try_ref_from_suffix_with_elems_dynamic_size: movabs rax, 4611686018427387901 cmp rdx, rax - ja .LBB5_1 - lea r8d, [rsi + rdi] - xor ecx, ecx - mov eax, 0 - test r8b, 1 - jne .LBB5_5 + ja .LBB5_3 lea rax, [2*rdx + 4] + lea ecx, [rsi + rdi] sub rsi, rax - jae .LBB5_4 -.LBB5_1: + setb al + or al, cl + test al, 1 + je .LBB5_4 +.LBB5_3: xor eax, eax - mov edx, 1 ret .LBB5_4: - lea r8, [rdi + rsi] - movzx esi, word ptr [rdi + rsi] - cmp si, -16192 - mov ecx, 2 - cmove rcx, rdx + lea rcx, [rdi + rsi] xor eax, eax - cmp esi, 49344 - cmove rax, r8 -.LBB5_5: - mov rdx, rcx + cmp word ptr [rdi + rsi], -16192 + cmove rax, rcx ret diff --git a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64.mca b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64.mca index d4f78f67a2..53ff061926 100644 --- a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64.mca +++ b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2400 -Total Cycles: 1107 -Total uOps: 2600 +Instructions: 1700 +Total Cycles: 544 +Total uOps: 1900 Dispatch Width: 4 -uOps Per Cycle: 2.35 -IPC: 2.17 -Block RThroughput: 6.5 +uOps Per Cycle: 3.49 +IPC: 3.13 +Block RThroughput: 4.8 Instruction Info: @@ -20,27 +20,20 @@ Instruction Info: [1] [2] [3] [4] [5] [6] Instructions: 1 1 0.33 movabs rax, 4611686018427387901 1 1 0.33 cmp rdx, rax - 1 1 1.00 ja .LBB5_1 - 1 1 0.50 lea r8d, [rsi + rdi] - 1 0 0.25 xor ecx, ecx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test r8b, 1 - 1 1 1.00 jne .LBB5_5 + 1 1 1.00 ja .LBB5_3 1 1 0.50 lea rax, [2*rdx + 4] + 1 1 0.50 lea ecx, [rsi + rdi] 1 1 0.33 sub rsi, rax - 1 1 1.00 jae .LBB5_4 + 1 1 0.50 setb al + 1 1 0.33 or al, cl + 1 1 0.33 test al, 1 + 1 1 1.00 je .LBB5_4 1 0 0.25 xor eax, eax - 1 1 0.33 mov edx, 1 1 1 1.00 U ret - 1 1 0.50 lea r8, [rdi + rsi] - 1 5 0.50 * movzx esi, word ptr [rdi + rsi] - 1 1 0.33 cmp si, -16192 - 1 1 0.33 mov ecx, 2 - 2 2 0.67 cmove rcx, rdx + 1 1 0.50 lea rcx, [rdi + rsi] 1 0 0.25 xor eax, eax - 1 1 0.33 cmp esi, 49344 - 2 2 0.67 cmove rax, r8 - 1 1 0.33 mov rdx, rcx + 2 6 0.50 * cmp word ptr [rdi + rsi], -16192 + 2 2 0.67 cmove rax, rcx 1 1 1.00 U ret @@ -57,31 +50,24 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.99 7.00 - 8.01 0.50 0.50 + - - 5.33 5.33 - 5.34 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.02 0.95 - 0.03 - - movabs rax, 4611686018427387901 - - - 0.93 0.04 - 0.03 - - cmp rdx, rax - - - - - - 1.00 - - ja .LBB5_1 - - - 0.96 0.04 - - - - lea r8d, [rsi + rdi] - - - - - - - - - xor ecx, ecx - - - 0.95 0.02 - 0.03 - - mov eax, 0 - - - 0.95 0.05 - - - - test r8b, 1 - - - - - - 1.00 - - jne .LBB5_5 - - - 0.06 0.94 - - - - lea rax, [2*rdx + 4] - - - 0.93 0.07 - - - - sub rsi, rax - - - - - - 1.00 - - jae .LBB5_4 + - - 0.66 0.33 - 0.01 - - movabs rax, 4611686018427387901 + - - 0.02 0.98 - - - - cmp rdx, rax + - - - - - 1.00 - - ja .LBB5_3 + - - 0.98 0.02 - - - - lea rax, [2*rdx + 4] + - - 0.66 0.34 - - - - lea ecx, [rsi + rdi] + - - 0.33 0.65 - 0.02 - - sub rsi, rax + - - 0.99 - - 0.01 - - setb al + - - 0.01 0.36 - 0.63 - - or al, cl + - - 0.01 0.96 - 0.03 - - test al, 1 + - - - - - 1.00 - - je .LBB5_4 - - - - - - - - xor eax, eax - - - 0.03 0.95 - 0.02 - - mov edx, 1 - - - - - 1.00 - - ret - - - 0.97 0.03 - - - - lea r8, [rdi + rsi] - - - - - - - 0.50 0.50 movzx esi, word ptr [rdi + rsi] - - - 0.03 0.97 - - - - cmp si, -16192 - - - 0.05 0.94 - 0.01 - - mov ecx, 2 - - - 0.06 0.98 - 0.96 - - cmove rcx, rdx + - - 0.32 0.68 - - - - lea rcx, [rdi + rsi] - - - - - - - - xor eax, eax - - - 0.97 0.03 - - - - cmp esi, 49344 - - - 0.06 0.96 - 0.98 - - cmove rax, r8 - - - 0.02 0.03 - 0.95 - - mov rdx, rcx + - - 0.36 0.02 - 0.62 0.50 0.50 cmp word ptr [rdi + rsi], -16192 + - - 0.99 0.99 - 0.02 - - cmove rax, rcx - - - - - 1.00 - - ret diff --git a/zerocopy/src/layout.rs b/zerocopy/src/layout.rs index d58786d637..a2ac64a1f2 100644 --- a/zerocopy/src/layout.rs +++ b/zerocopy/src/layout.rs @@ -698,41 +698,31 @@ impl DstLayout { None => return Err(MetadataCastError::Size), }; - // Calculate the number of elements that fit in - // `max_slice_and_padding_bytes`; any remaining bytes will be - // considered padding. + // Calculate the maximum number of elements that fit in + // `max_slice_and_padding_bytes`. // // Guaranteed not to divide by zero: `elem_size` is non-zero. #[allow(clippy::arithmetic_side_effects)] let elems = max_slice_and_padding_bytes / elem_size.get(); - // Guaranteed not to overflow on multiplication: `usize::MAX >= - // max_slice_and_padding_bytes >= (max_slice_and_padding_bytes / - // elem_size) * elem_size`. + // Let M = max_total_bytes, A = self.align, E = elem_size, and + // r = (M - offset) % E. Division gives M - offset = elems * E + // + r, so the unpadded size is offset + elems * E = M - r. + // M is a multiple of A. When E <= A, r < E <= A, so rounding + // M - r up to A gives M without reconstructing it from `elems`. // - // Guaranteed not to overflow on addition: - // - max_slice_and_padding_bytes == max_total_bytes - offset - // - elems * elem_size <= max_slice_and_padding_bytes == max_total_bytes - offset - // - elems * elem_size + offset <= max_total_bytes <= usize::MAX + // For E > A, keep the multiplication: computing the size from + // the remainder can cause LLVM to duplicate division work. #[allow(clippy::arithmetic_side_effects)] - let without_padding = offset + elems * elem_size.get(); - // `self_bytes` is equal to the offset bytes plus the bytes - // consumed by the trailing slice plus any padding bytes - // required to satisfy the alignment. Note that we have computed - // the maximum number of trailing slice elements that could fit - // in `self_bytes`, so any padding is guaranteed to be less than - // the size of an extra element. - // - // Guaranteed not to overflow: - // - By previous comment: without_padding == elems * elem_size + - // offset <= max_total_bytes - // - By construction, `max_total_bytes` is a multiple of - // `self.align`. - // - At most, adding padding needed to round `without_padding` - // up to the next multiple of the alignment will bring - // `self_bytes` up to `max_total_bytes`. - #[allow(clippy::arithmetic_side_effects)] - let self_bytes = - without_padding + util::padding_needed_for(without_padding, self.align); + let self_bytes = if elem_size.get() <= self.align.get() { + max_total_bytes + } else { + // Neither operation can overflow: elems * E <= M - offset, + // so adding offset gives a value <= M <= usize::MAX. + let without_padding = offset + elems * elem_size.get(); + // Rounding up cannot overflow either: M is a multiple of + // A, so rounding a value <= M up to A still gives <= M. + without_padding + util::padding_needed_for(without_padding, self.align) + }; (elems, self_bytes) } }; @@ -745,7 +735,7 @@ impl DstLayout { // - In the `Sized` branch, only returns `size` if `size <= // bytes_len`. // - In the `SliceDst` branch, calculates `self_bytes <= - // max_toatl_bytes`, which is upper-bounded by `bytes_len`. + // max_total_bytes`, which is upper-bounded by `bytes_len`. #[allow(clippy::arithmetic_side_effects)] CastType::Suffix => bytes_len - self_bytes, }; @@ -1513,6 +1503,9 @@ mod tests { fn validate_behavior( (layout, addr, bytes_len, cast_type): (DstLayout, usize, usize, CastType), ) { + use core::convert::TryFrom as _; + + let wide = |n: usize| u128::try_from(n).unwrap(); if let Ok((elems, split_at)) = layout.validate_cast_and_convert_metadata(addr, bytes_len, cast_type) { @@ -1530,32 +1523,40 @@ mod tests { assert!(!(sized && elems != 0), "{}", debug_str); let resulting_size = match layout.size_info { - SizeInfo::Sized { size } => size, + SizeInfo::Sized { size } => wide(size), SizeInfo::SliceDst(TrailingSliceLayout { offset, elem_size }) => { - let padded_size = |elems| { - let without_padding = offset + elems * elem_size; - without_padding + util::padding_needed_for(without_padding, align) + // Use wider arithmetic so even the next element's + // padded size can exceed `usize::MAX` without wrapping. + let padded_size = |elems: u128| { + let without_padding = wide(offset) + elems * wide(elem_size); + let align = wide(align.get()); + ((without_padding + align - 1) / align) * align }; - let resulting_size = padded_size(elems); + let resulting_size = padded_size(wide(elems)); // Test that `validate_cast_and_convert_metadata` // computed the largest possible value that fits in the // given range. - assert!(padded_size(elems + 1) > bytes_len, "{}", debug_str); + assert!(padded_size(wide(elems) + 1) > wide(bytes_len), "{}", debug_str); resulting_size } }; // Test safety postconditions guaranteed by // `validate_cast_and_convert_metadata`. - assert!(resulting_size <= bytes_len, "{}", debug_str); + assert!(resulting_size <= wide(bytes_len), "{}", debug_str); match cast_type { CastType::Prefix => { assert_eq!(addr % align, 0, "{}", debug_str); - assert_eq!(resulting_size, split_at, "{}", debug_str); + assert_eq!(resulting_size, wide(split_at), "{}", debug_str); } CastType::Suffix => { - assert_eq!(split_at, bytes_len - resulting_size, "{}", debug_str); + assert_eq!( + wide(split_at), + wide(bytes_len) - resulting_size, + "{}", + debug_str + ); assert_eq!((addr + split_at) % align, 0, "{}", debug_str); } } @@ -1592,6 +1593,42 @@ mod tests { .map(|(size_info, align)| layout(size_info, align)); itertools::iproduct!(layouts, 0..8, 0..8, [CastType::Prefix, CastType::Suffix]) .for_each(validate_behavior); + + // Exercise remainders below, at, and above the alignment, as well as + // sizes near integer limits. No allocation is needed: this method + // checks layout arithmetic independently of pointer validity. + let large = DstLayout::MAX_SIZE - 32; + let layouts = itertools::iproduct!( + [0, 1, 3, 8, large], + [1, 3, 4, 12, 17, large], + [1, 2, 4, 8, 16, DstLayout::CURRENT_MAX_ALIGN.get()] + ) + .map(|(offset, elem_size, align)| layout((offset, elem_size), align)); + let lengths = [ + 0, + 3, + 4, + 7, + 8, + 15, + 16, + 17, + 31, + 32, + 33, + large, + usize::MAX - 31, + usize::MAX - 1, + usize::MAX, + ]; + itertools::iproduct!( + layouts, + [0usize, 1, 31], + lengths, + [CastType::Prefix, CastType::Suffix] + ) + .filter(|(_, addr, len, _)| addr.checked_add(*len).is_some()) + .for_each(validate_behavior); } #[test] diff --git a/zerocopy/src/lib.rs b/zerocopy/src/lib.rs index 3ac6db76fd..944f177c0b 100644 --- a/zerocopy/src/lib.rs +++ b/zerocopy/src/lib.rs @@ -1723,10 +1723,10 @@ runtime checks using `#[zerocopy(invariant(expression))]`: #[derive(TryFromBytes)] struct Foo { a: u8, - #[zerocopy(invariant((**a.unaligned_as_ref() % 2) == (**b.unaligned_as_ref() as u8)))] + #[zerocopy(invariant((*a.read() % 2) == (*b.read() as u8)))] b: bool, - #[zerocopy(invariant(**c.unaligned_as_ref() > 0))] - c: i8, + #[zerocopy(invariant(*c.read() > 0))] + c: i16, } ``` @@ -1734,7 +1734,9 @@ Each expression must return a `bool`. It has access to validated, read-only [`Ptr`]s to the current field and all preceding fields of the struct or variant, using their field names. A union's invariants have access only to the current field. The expression can use the existing [`Ptr`] APIs to -inspect those fields. In this example, all fields are accessed by reference. +inspect those fields. In this example, `read()` copies each field without +requiring alignment, and dereferencing the resulting [`ReadOnly`] accesses +the copied value. Rust's usual restrictions on local bindings apply; for example, a field name cannot shadow an in-scope constant. @@ -3281,17 +3283,7 @@ pub unsafe trait TryFromBytes { where Self: Sized, { - // FIXME(#2981): If `align_of::() == 1`, validate `source` in-place. - - let candidate = match CoreMaybeUninit::::read_from_bytes(source) { - Ok(candidate) => candidate, - Err(e) => { - return Err(TryReadError::Size(e.with_dst())); - } - }; - // SAFETY: `candidate` was copied from from `source: &[u8]`, so all of - // its bytes are initialized. - unsafe { try_read_from(source, candidate) } + try_read_from(source) } /// Attempts to read a `Self` from the prefix of the given `source`. @@ -3359,17 +3351,17 @@ pub unsafe trait TryFromBytes { where Self: Sized, { - // FIXME(#2981): If `align_of::() == 1`, validate `source` in-place. - - let (candidate, suffix) = match CoreMaybeUninit::::read_from_prefix(source) { - Ok(candidate) => candidate, - Err(e) => { - return Err(TryReadError::Size(e.with_dst())); - } + let (prefix, suffix) = match SplitAt::split_at(source, mem::size_of::()) { + Some(split) => split.via_immutable(), + None => return Err(SizeError::new(source).into()), }; - // SAFETY: `candidate` was copied from from `source: &[u8]`, so all of - // its bytes are initialized. - unsafe { try_read_from(source, candidate).map(|slf| (slf, suffix)) } + match try_read_from(prefix) { + Ok(slf) => Ok((slf, suffix)), + Err(e) => Err(e.map_src( + #[inline(always)] + |_| source, + )), + } } /// Attempts to read a `Self` from the suffix of the given `source`. @@ -3438,37 +3430,65 @@ pub unsafe trait TryFromBytes { where Self: Sized, { - // FIXME(#2981): If `align_of::() == 1`, validate `source` in-place. - - let (prefix, candidate) = match CoreMaybeUninit::::read_from_suffix(source) { - Ok(candidate) => candidate, - Err(e) => { - return Err(TryReadError::Size(e.with_dst())); - } + let split_at = match source.len().checked_sub(mem::size_of::()) { + Some(split_at) => split_at, + None => return Err(SizeError::new(source).into()), }; - // SAFETY: `candidate` was copied from from `source: &[u8]`, so all of - // its bytes are initialized. - unsafe { try_read_from(source, candidate).map(|slf| (prefix, slf)) } + // SAFETY: `checked_sub` returned the difference without overflow [1], + // so `split_at = source.len() - size_of::() <= source.len()`. + // This satisfies `SplitAt::split_at_unchecked`'s precondition. + // + // [1] Per https://doc.rust-lang.org/1.56.0/std/primitive.usize.html#method.checked_sub: + // + // Checked integer subtraction. Computes `self - rhs`, returning + // `None` if overflow occurred. + let (prefix, suffix) = + unsafe { SplitAt::split_at_unchecked(source, split_at) }.via_immutable(); + match try_read_from(suffix) { + Ok(slf) => Ok((prefix, slf)), + Err(e) => Err(e.map_src( + #[inline(always)] + |_| source, + )), + } } } +/// Validates and interprets the given affix of `source`'s bytes as a `&T`. +/// +/// Returns the destination reference and excess bytes on success. All errors, +/// including validity errors, contain the original `&S`. #[inline(always)] -fn try_ref_from_prefix_suffix( - source: &[u8], +fn try_ref_from_prefix_suffix( + source: &S, cast_type: CastType, meta: Option, -) -> Result<(&T, &[u8]), TryCastError<&[u8], T>> { - match Ptr::from_ref(source).try_cast_into::(cast_type, meta) { - Ok((source, prefix_suffix)) => { +) -> Result<(&T, &[u8]), TryCastError<&S, T>> +where + S: IntoBytes + Immutable + ?Sized, + T: TryFromBytes + KnownLayout + Immutable + ?Sized, +{ + match Ptr::from_ref(source.as_bytes()).try_cast_into::(cast_type, meta) { + Ok((candidate, prefix_suffix)) => { // This call may panic. If that happens, it doesn't cause any soundness // issues, as we have not generated any invalid state which we need to // fix before returning. - match source.try_into_safe() { + match candidate.try_into_safe() { Ok(valid) => Ok((valid.as_ref(), prefix_suffix.as_ref())), - Err(e) => Err(e.map_src(|src| src.as_bytes::().as_ref()).into()), + Err(e) => Err(e + .map_src( + #[inline(always)] + |_| source, + ) + .into()), } } - Err(e) => Err(e.map_src(Ptr::as_ref).into()), + Err(e) => Err(e + .map_src( + #[inline(always)] + |_| source, + ) + .into()), } } @@ -3497,49 +3517,112 @@ fn swap((t, u): (T, U)) -> (U, T) { (u, t) } -/// # Safety -/// -/// All bytes of `candidate` must be initialized. #[inline(always)] -unsafe fn try_read_from( - source: S, - mut candidate: CoreMaybeUninit, -) -> Result> { +fn try_read_from(source: &S) -> Result> +where + S: IntoBytes + Immutable + ?Sized, + T: TryFromBytes, +{ + let bytes = source.as_bytes(); + + if bytes.len() != mem::size_of::() { + return Err(SizeError::new(source).into()); + } + + // FIXME(#2981): Avoid the validation copy when validation can safely use + // a pointer derived from `source`'s shared borrow. + + // Initialize the candidate in its final location. A typed move of a + // `MaybeUninit` may discard initialized bytes at `T`'s padding offsets + // [1], but validation requires every byte to be initialized. Do not move + // `candidate` between this copy and validation. + // + // [1] Per https://doc.rust-lang.org/1.93.1/std/mem/union.MaybeUninit.html#validity: + // + // Moving or copying a value of type `MaybeUninit` (i.e., performing a + // "typed copy") will exactly preserve the contents, including the + // provenance, of all non-padding bytes of type `T` in the value's + // representation. + let mut candidate = CoreMaybeUninit::::uninit(); + + // SAFETY: The earlier `if bytes.len() != mem::size_of::()` returns on a + // size mismatch, so reaching this copy implies that `bytes.len()` equals + // `mem::size_of::()`. Copying that many `u8`s cannot overrun `bytes`. + // `candidate.as_mut_ptr()` is writable for the same number of bytes because + // `MaybeUninit` has `T`'s size. Both pointers are non-null and aligned + // for `u8`, including when `T` is zero-sized. The fresh local allocation + // cannot overlap `source`, which is an argument. Thus the read, write, + // alignment, and non-overlap requirements of [2] hold. Copying `u8`s + // initializes every destination byte, including those at `T`'s padding + // offsets. + // + // These writes cannot violate `candidate`'s bit validity because every bit + // pattern is valid for `MaybeUninit` [3], even if it is invalid for `T`. + // + // [2] Per https://doc.rust-lang.org/1.56.0/std/ptr/fn.copy_nonoverlapping.html: + // + // Copies `count * size_of::()` bytes from `src` to `dst`. The source + // and destination must *not* overlap. + // + // [3] Per https://doc.rust-lang.org/1.56.0/std/mem/union.MaybeUninit.html#layout: + // + // ... any bit value is valid for a `MaybeUninit` ... + unsafe { + ptr::copy_nonoverlapping( + bytes.as_ptr(), + candidate.as_mut_ptr().cast::(), + mem::size_of::(), + ); + } + // We use `from_mut` despite not mutating via `c_ptr` so that we don't need // to add a `T: Immutable` bound. let c_ptr = Ptr::from_mut(&mut candidate); + // SAFETY: `c_ptr` has no uninitialized sub-ranges because it derived from - // `candidate`, which the caller promises is entirely initialized. Since - // `candidate` is a `MaybeUninit`, it has no validity requirements, and so - // no values written to an `Initialized` `c_ptr` can violate its validity. - // Since `c_ptr` has `Exclusive` aliasing, no mutations may happen except - // via `c_ptr` so long as it is live, so we don't need to worry about the - // fact that `c_ptr` may have more restricted validity than `candidate`. + // `candidate`, whose bytes were all initialized by the copy above and which + // has not been moved since. Since `candidate` is a `MaybeUninit`, it has no + // validity requirements, and so no values written to an `Initialized` + // `c_ptr` can violate its validity. Since `c_ptr` has `Exclusive` aliasing, + // no mutations may happen except via `c_ptr` so long as it is live, so we + // don't need to worry about the fact that `c_ptr` may have more restricted + // validity than `candidate`. let c_ptr = unsafe { c_ptr.assume_validity::() }; - let mut c_ptr = c_ptr.cast::<_, crate::pointer::cast::CastSized, _>(); - // Since we don't have `T: KnownLayout`, we hack around that by using - // `Wrapping`, which implements `KnownLayout` even if `T` doesn't. + let c_ptr = c_ptr.cast::<_, crate::pointer::cast::CastSized, _>(); + + // SAFETY: `c_ptr` originated from a reference to `candidate`, so its + // address is aligned for `MaybeUninit`, which has `T`'s alignment [1]. + // `CastSized` preserves that address. `ReadOnly` is `repr(transparent)` + // with a single `T` field, so it also has `T`'s alignment [2], including + // when `T` is zero-sized. + // + // [1] Per https://doc.rust-lang.org/1.56.0/std/mem/union.MaybeUninit.html#layout: // + // `MaybeUninit` is guaranteed to have the same size, alignment, and + // ABI as `T`: + // + // [2] Per https://doc.rust-lang.org/1.93.1/reference/type-layout.html#the-transparent-representation: + // + // ... same layout and ABI as the only non-size 0 non-alignment 1 field, + // if present, or unit otherwise. + let mut c_ptr = unsafe { c_ptr.assume_alignment::() }; + // This call may panic. If that happens, it doesn't cause any soundness // issues, as we have not generated any invalid state which we need to fix // before returning. - if !Wrapping::::is_safe(c_ptr.reborrow_shared().forget_aligned()) { + if !T::is_safe(c_ptr.reborrow_shared()) { return Err(ValidityError::new(source).into()); } - fn _assert_same_size_and_validity() - where - Wrapping: pointer::TransmuteFrom, - T: pointer::TransmuteFrom, invariant::Safe, invariant::Safe>, - { - } - - _assert_same_size_and_validity::(); - - // SAFETY: We just validated that `candidate` contains a valid - // `Wrapping`, which has the same size and bit validity as `T`, as - // guaranteed by the preceding type assertion. + // SAFETY: `T::is_safe` returned true for `candidate`'s initialized bytes, + // so it contains a valid `T`, as required by [1]. Validation used a shared + // `ReadOnly` pointer, and the bytes have not been modified since. + // + // [1] Per https://doc.rust-lang.org/1.56.0/std/mem/union.MaybeUninit.html#method.assume_init: + // + // It is up to the caller to guarantee that the `MaybeUninit` really is + // in an initialized state. Ok(unsafe { candidate.assume_init() }) } @@ -5622,22 +5705,32 @@ pub unsafe trait FromBytes: FromZeros { } } -/// Interprets the given affix of the given bytes as a `&Self`. +/// Interprets the given affix of `source`'s bytes as a `&T`. /// -/// This method computes the largest possible size of `Self` that can fit in the -/// prefix or suffix bytes of `source`, then attempts to return both a reference -/// to those bytes interpreted as a `Self`, and a reference to the excess bytes. -/// If there are insufficient bytes, or if that affix of `source` is not -/// appropriately aligned, this returns `Err`. +/// This method uses `meta` if provided; otherwise, it computes the largest +/// possible size of `T` that can fit in the prefix or suffix bytes of `source`. +/// It returns both a reference to those bytes interpreted as a `T`, and a +/// reference to the excess bytes. If there are insufficient bytes, or if that +/// affix of `source` is not appropriately aligned, this returns `Err` containing +/// the original `&S`. #[inline(always)] -fn ref_from_prefix_suffix( - source: &[u8], +fn ref_from_prefix_suffix( + source: &S, meta: Option, cast_type: CastType, -) -> Result<(&T, &[u8]), CastError<&[u8], T>> { - let (slf, prefix_suffix) = Ptr::from_ref(source) +) -> Result<(&T, &[u8]), CastError<&S, T>> +where + S: IntoBytes + Immutable + ?Sized, + T: FromBytes + KnownLayout + Immutable + ?Sized, +{ + let (slf, prefix_suffix) = Ptr::from_ref(source.as_bytes()) .try_cast_into::<_, BecauseImmutable>(cast_type, meta) - .map_err(|err| err.map_src(|s| s.as_ref()))?; + .map_err(|err| { + err.map_src( + #[inline(always)] + |_| source, + ) + })?; Ok((slf.recall_validity().as_ref(), prefix_suffix.as_ref())) } @@ -7366,6 +7459,112 @@ mod tests { ); } + #[test] + fn test_try_read_from_typed_source() { + #[derive(IntoBytes, Immutable)] + #[repr(transparent)] + struct Source(T); + + fn check(source: &S, valid: bool) { + match try_read_from::<_, [bool; 4]>(source) { + Ok(value) => { + assert!(valid); + assert_eq!(value, [false, true, false, true]); + } + Err(error) => { + assert!(!valid); + assert!(matches!(error, TryReadError::Validity(_))); + assert!(ptr::eq(error.into_src(), source)); + } + } + + let error = try_read_from::<_, [bool; 5]>(source).unwrap_err(); + assert!(matches!(error, TryReadError::Size(_))); + assert!(ptr::eq(error.into_src(), source)); + } + + for (value, valid) in [(u16::from_ne_bytes([0, 1]), true), (0x0202, false)] { + let source = Source([value; 2]); + check(&source, valid); + let source: &Source<[u16]> = &source; + check(source, valid); + } + + let source = Source([(); 2]); + assert!(try_read_from::<_, ()>(&source).is_ok()); + let source: &Source<[()]> = &source; + assert!(try_read_from::<_, ()>(source).is_ok()); + } + + #[test] + fn test_try_read_initialized_padding() { + #[derive(KnownLayout, Immutable, Debug, PartialEq, Eq)] + #[repr(C, align(8))] + struct Padded(u8); + + // SAFETY: `Padded` has only a `u8` field and padding, so all + // initialized byte sequences are valid. Per + // https://doc.rust-lang.org/1.93.1/reference/behavior-considered-undefined.html#invalid-values: + // + // An integer (`i*`/`u*`), floating point value (`f*`), or raw pointer + // must be initialized, i.e., must not be obtained from uninitialized + // memory. + // + // A `struct`, tuple, and array requires all fields/elements to be + // valid at their respective type. + const _: () = unsafe { + unsafe_impl!(=> TryFromBytes for Padded; |candidate| { + // Observe every byte while validation is running, including + // padding that a typed move of `MaybeUninit` could + // discard. The returned `Padded` need not retain its padding. + assert_eq!(candidate.as_bytes::().as_ref(), &[0xA5; 8]); + true + }) + }; + + let source = [0xA5; 8]; + assert_eq!(Padded::try_read_from_bytes(&source), Ok(Padded(0xA5))); + + let prefix_source = [0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0]; + let (value, suffix) = Padded::try_read_from_prefix(&prefix_source).unwrap(); + assert_eq!(value, Padded(0xA5)); + assert!(ptr::eq(suffix, &prefix_source[8..])); + + let suffix_source = [0, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5]; + let (prefix, value) = Padded::try_read_from_suffix(&suffix_source).unwrap(); + assert_eq!(value, Padded(0xA5)); + assert!(ptr::eq(prefix, &suffix_source[..1])); + } + + #[test] + fn test_try_read_error_sources_and_zero_sized() { + let invalid = [2, 3]; + for (error, source) in [ + (bool::try_read_from_bytes(&invalid[..1]).unwrap_err(), &invalid[..1]), + (bool::try_read_from_prefix(&invalid).unwrap_err(), &invalid[..]), + (bool::try_read_from_suffix(&invalid).unwrap_err(), &invalid[..]), + ] { + assert!(matches!(error, TryReadError::Validity(_))); + assert!(ptr::eq(error.into_src(), source)); + } + for source in [&[][..], &invalid[..]] { + let error = bool::try_read_from_bytes(source).unwrap_err(); + assert!(matches!(error, TryReadError::Size(_))); + assert!(ptr::eq(error.into_src(), source)); + } + for error in [ + bool::try_read_from_prefix(&invalid[..0]).unwrap_err(), + bool::try_read_from_suffix(&invalid[..0]).unwrap_err(), + ] { + assert!(matches!(error, TryReadError::Size(_))); + assert!(ptr::eq(error.into_src(), &invalid[..0])); + } + + assert_eq!(<()>::try_read_from_bytes(&[]), Ok(())); + assert_eq!(<()>::try_read_from_prefix(&invalid), Ok(((), &invalid[..]))); + assert_eq!(<()>::try_read_from_suffix(&invalid), Ok((&invalid[..], ()))); + } + #[test] fn test_ref_from_mut_from_bytes() { // Test `FromBytes::{ref_from_bytes, mut_from_bytes}{,_prefix,Suffix}` @@ -7785,6 +7984,133 @@ mod tests { assert_eq!(rest.len(), 4); } + #[test] + fn test_ref_from_prefix_suffix_typed_source() { + #[derive(IntoBytes, Immutable)] + #[repr(transparent)] + struct Source(T); + + fn check(source: &S) { + let bytes = source.as_bytes(); + assert_eq!(bytes.len(), 4); + + for cast_type in [CastType::Prefix, CastType::Suffix] { + let (expected, expected_rest) = match cast_type { + CastType::Prefix => (&bytes[..3], &bytes[3..]), + CastType::Suffix => (&bytes[1..], &bytes[..1]), + }; + for meta in [None, Some(1)] { + let (value, rest) = + ref_from_prefix_suffix::<_, [[u8; 3]]>(source, meta, cast_type).unwrap(); + assert_eq!(value, &[expected]); + assert!(ptr::eq(value.as_bytes(), expected)); + assert!(ptr::eq(rest, expected_rest)); + } + + let err = + ref_from_prefix_suffix::<_, [u8; 5]>(source, None, cast_type).unwrap_err(); + assert!(matches!(err, CastError::Size(_))); + assert!(ptr::eq(err.into_src(), source)); + + let err = + ref_from_prefix_suffix::<_, [u8]>(source, Some(5), cast_type).unwrap_err(); + assert!(matches!(err, CastError::Size(_))); + assert!(ptr::eq(err.into_src(), source)); + } + } + + check(&Source([0x0102u16, 0x0304])); + let source = Source([false, true, false, true]); + let source: &Source<[bool]> = &source; + check(source); + } + + #[test] + fn test_ref_from_prefix_suffix_typed_source_alignment_error() { + let storage = Align::<[bool; 9], AU64>::new([false; 9]); + for cast_type in [CastType::Prefix, CastType::Suffix] { + let source = match cast_type { + CastType::Prefix => &storage.t[1..], + CastType::Suffix => &storage.t[..], + }; + let err = ref_from_prefix_suffix::<_, AU64>(source, None, cast_type).unwrap_err(); + assert!(matches!(err, CastError::Alignment(_))); + let recovered: &[bool] = err.into_src(); + assert!(ptr::eq(recovered, source)); + + let err = try_ref_from_prefix_suffix::<_, AU64>(source, cast_type, None).unwrap_err(); + assert!(matches!(err, TryCastError::Alignment(_))); + let recovered: &[bool] = err.into_src(); + assert!(ptr::eq(recovered, source)); + } + } + + #[test] + fn test_try_ref_from_prefix_suffix_typed_source() { + #[derive(IntoBytes, Immutable)] + #[repr(transparent)] + struct Source(T); + + fn check(source: &S) { + let bytes = source.as_bytes(); + assert_eq!(bytes, [0, 1, 0, 1]); + + for cast_type in [CastType::Prefix, CastType::Suffix] { + let (expected, expected_rest) = match cast_type { + CastType::Prefix => (&bytes[..3], &bytes[3..]), + CastType::Suffix => (&bytes[1..], &bytes[..1]), + }; + for meta in [None, Some(1)] { + let (value, rest) = + try_ref_from_prefix_suffix::<_, [[bool; 3]]>(source, cast_type, meta) + .unwrap(); + assert_eq!(value.as_bytes(), expected); + assert!(ptr::eq(value.as_bytes(), expected)); + assert!(ptr::eq(rest, expected_rest)); + } + + let err = try_ref_from_prefix_suffix::<_, [bool; 5]>(source, cast_type, None) + .unwrap_err(); + assert!(matches!(err, TryCastError::Size(_))); + assert!(ptr::eq(err.into_src(), source)); + + let err = try_ref_from_prefix_suffix::<_, [bool]>(source, cast_type, Some(5)) + .unwrap_err(); + assert!(matches!(err, TryCastError::Size(_))); + assert!(ptr::eq(err.into_src(), source)); + } + } + + check(&Source([u16::from_ne_bytes([0, 1]); 2])); + let source = Source([false, true, false, true]); + let source: &Source<[bool]> = &source; + check(source); + } + + #[test] + fn test_try_ref_from_prefix_suffix_typed_source_validity_error() { + fn check(source: &S) { + for cast_type in [CastType::Prefix, CastType::Suffix] { + let err = + try_ref_from_prefix_suffix::<_, bool>(source, cast_type, None).unwrap_err(); + assert!(matches!(err, TryCastError::Validity(_))); + assert!(ptr::eq(err.into_src(), source)); + + for meta in [None, Some(1)] { + let err = try_ref_from_prefix_suffix::<_, [bool]>(source, cast_type, meta) + .unwrap_err(); + assert!(matches!(err, TryCastError::Validity(_))); + assert!(ptr::eq(err.into_src(), source)); + } + } + } + + let source = [u16::from_ne_bytes([2, 2]); 2]; + check(&source); + check(&source[..]); + check(source.as_bytes()); + } + #[test] fn test_try_ref_from_prefix_suffix() { use crate::util::testutil::Align; diff --git a/zerocopy/src/wrappers.rs b/zerocopy/src/wrappers.rs index 4425ec26cb..2d4ff55159 100644 --- a/zerocopy/src/wrappers.rs +++ b/zerocopy/src/wrappers.rs @@ -8,7 +8,7 @@ // This file may not be copied, modified, or distributed except according to // those terms. -use core::{fmt, hash::Hash}; +use core::{borrow::Borrow, fmt, hash::Hash}; use super::*; use crate::pointer::{invariant::Safe, SizeEq, TransmuteFrom}; @@ -613,7 +613,14 @@ mod read_only_def { /// Note that `&mut ReadOnly` still permits mutation – the read-only /// property only applies to shared references. /// + /// `ReadOnly` implements [`Copy`] and [`Clone`] when `T: Copy`. Cloning + /// copies the value without calling `T::clone`. Trait implementations that + /// expose or operate on a shared reference to `T` require `T: Immutable`. + /// Constructing a wrapper or accessing its contents through [`AsMut`] + /// does not require `T: Immutable`. + /// /// [`Immutable`]: crate::Immutable + #[derive(Copy)] #[repr(transparent)] pub struct ReadOnly { // INVARIANT: `inner` is never mutated through a `&ReadOnly` @@ -715,6 +722,29 @@ unsafe impl TransmuteFrom for ReadOnly {} // it has the same bit validity as `T`. unsafe impl TransmuteFrom, Safe, Safe> for T {} +impl Default for ReadOnly { + #[inline(always)] + fn default() -> Self { + Self::new(Default::default()) + } +} + +// Copying avoids calling `T::clone`, which could mutate through a shared +// reference to `T` and violate `ReadOnly`'s invariant. +impl Clone for ReadOnly { + #[inline(always)] + fn clone(&self) -> Self { + *self + } +} + +impl From for ReadOnly { + #[inline(always)] + fn from(t: T) -> Self { + Self::new(t) + } +} + impl<'a, T: ?Sized + Immutable> From<&'a T> for &'a ReadOnly { #[inline(always)] fn from(t: &'a T) -> &'a ReadOnly { @@ -743,13 +773,72 @@ impl DerefMut for ReadOnly { } } -impl Debug for ReadOnly { +impl AsRef for ReadOnly { #[inline(always)] - fn fmt(&self, f: &mut Formatter<'_>) -> fmt::Result { - self.deref().fmt(f) + fn as_ref(&self) -> &T { + self.deref() + } +} + +impl Borrow for ReadOnly { + #[inline(always)] + fn borrow(&self) -> &T { + self.deref() + } +} + +impl AsMut for ReadOnly { + #[inline(always)] + fn as_mut(&mut self) -> &mut T { + ReadOnly::as_mut(self) + } +} + +// Delegate shared access through `Deref`, which requires `T: Immutable` and +// thus prevents interior mutation of the wrapped value by these trait methods. +impl PartialEq for ReadOnly { + #[inline(always)] + fn eq(&self, other: &Self) -> bool { + self.deref().eq(other.deref()) + } +} + +impl Eq for ReadOnly {} + +impl PartialOrd for ReadOnly { + #[inline(always)] + fn partial_cmp(&self, other: &Self) -> Option { + self.deref().partial_cmp(other.deref()) } } +impl Ord for ReadOnly { + #[inline(always)] + fn cmp(&self, other: &Self) -> Ordering { + self.deref().cmp(other.deref()) + } +} + +impl Hash for ReadOnly { + #[inline(always)] + fn hash(&self, state: &mut H) { + self.deref().hash(state); + } +} + +macro_rules! impl_read_only_fmt { + ($($trait:ident),* $(,)?) => {$( + impl fmt::$trait for ReadOnly { + #[inline(always)] + fn fmt(&self, f: &mut Formatter<'_>) -> fmt::Result { + fmt::$trait::fmt(self.deref(), f) + } + } + )*}; +} + +impl_read_only_fmt!(Debug, Display, Binary, Octal, LowerHex, UpperHex, LowerExp, UpperExp); + // SAFETY: See safety comment on `ProjectToTag`. unsafe impl + ?Sized, Client> HasTag for ReadOnly { #[allow(clippy::missing_inline_in_public_items)] @@ -857,6 +946,96 @@ mod tests { use super::*; use crate::util::testutil::*; + #[test] + #[allow(clippy::clone_on_copy, clippy::non_canonical_clone_impl)] + fn test_read_only_copy_clone() { + // This type deliberately does not implement `Immutable`. Copying the + // wrapper must not require it or invoke the inner `Clone` impl. + #[derive(Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] + struct PanickingClone(u16); + + impl Clone for PanickingClone { + fn clone(&self) -> Self { + panic!("ReadOnly must not call the inner Clone implementation") + } + } + + static_assertions::assert_not_impl_any!( + ReadOnly: Debug, PartialEq, Eq, PartialOrd, Ord, + Hash, AsRef, Borrow + ); + static_assertions::assert_not_impl_any!( + ReadOnly>: Copy, Clone, Deref, AsRef>, Borrow> + ); + + let value = ReadOnly::new(PanickingClone(42)); + let copy = value; + let clone = value.clone(); + assert_eq!(ReadOnly::into_inner(copy).0, 42); + assert_eq!(ReadOnly::into_inner(clone).0, 42); + } + + #[test] + fn test_read_only_mutable_access() { + // Construction and exclusive access also support interior mutability. + let mut value = ReadOnly::>::default(); + let inner: &mut Cell = value.as_mut(); + inner.set(42); + assert_eq!(ReadOnly::into_inner(value).get(), 42); + + let value = ReadOnly::from(Cell::new(7)); + assert_eq!(ReadOnly::into_inner(value).get(), 7); + } + + #[test] + fn test_read_only_borrowed_lookup() { + // Borrowed lookups require equality, hashing, and ordering to agree + // between the wrapper and the inner value. + let mut hash_map = std::collections::HashMap::new(); + hash_map.insert(ReadOnly::new(42u16), true); + assert_eq!(hash_map.get(&42u16), Some(&true)); + assert_eq!(hash_map.get(&43u16), None); + + let mut tree = std::collections::BTreeMap::new(); + tree.insert(ReadOnly::new(42u16), true); + assert_eq!(tree.get(&42u16), Some(&true)); + assert_eq!(tree.get(&43u16), None); + + // Preserve partial orders, including incomparable values. + let nan = ReadOnly::new(f32::NAN); + assert_eq!(nan.partial_cmp(&nan), None); + assert_ne!(nan, nan); + } + + #[test] + fn test_read_only_unsized() { + let bytes = [1u8, 2]; + let value: &ReadOnly<[u8]> = (&bytes[..]).into(); + let as_ref: &[u8] = value.as_ref(); + let borrowed: &[u8] = value.borrow(); + assert_eq!(as_ref, bytes); + assert_eq!(borrowed, bytes); + assert_eq!(value, value); + assert_eq!(value.cmp(value), Ordering::Equal); + + let text: &ReadOnly = "hello".into(); + assert_eq!(format!("{:.3}", text), "hel"); + } + + #[test] + fn test_read_only_formatting() { + let value = ReadOnly::new(42u16); + // Check that formatting flags are forwarded as well as the value. + assert_eq!(format!("{:04?}", value), "0042"); + assert_eq!(format!("{:04}", value), "0042"); + assert_eq!(format!("{:#010b}", value), "0b00101010"); + assert_eq!(format!("{:#06o}", value), "0o0052"); + assert_eq!(format!("{:#06x}", value), "0x002a"); + assert_eq!(format!("{:#06X}", value), "0x002A"); + assert_eq!(format!("{:.2e}", value), "4.20e1"); + assert_eq!(format!("{:.2E}", value), "4.20E1"); + } + #[test] fn test_unalign() { // Test methods that don't depend on alignment. diff --git a/zerocopy/zerocopy-derive/tests/invariant.rs b/zerocopy/zerocopy-derive/tests/invariant.rs index e4a63ca24d..12bf65ace3 100644 --- a/zerocopy/zerocopy-derive/tests/invariant.rs +++ b/zerocopy/zerocopy-derive/tests/invariant.rs @@ -30,9 +30,9 @@ fn read( #[repr(C, packed)] struct Foo { a: u8, - #[zerocopy(invariant((read(a) % 2) == (read(b) as u8)))] + #[zerocopy(invariant((*a.read() % 2) == (*b.read() as u8)))] b: bool, - #[zerocopy(invariant(read(c) > 0))] + #[zerocopy(invariant(*c.read() > 0))] c: i16, }