diff --git a/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64 b/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64 index a58592a245..f380fee6c2 100644 --- a/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64 +++ b/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64 @@ -1,16 +1,16 @@ bench_ref_from_prefix_dynamic_padding: - xor edx, edx - mov eax, 0 test dil, 3 - je .LBB5_1 - ret -.LBB5_1: + jne .LBB5_2 movabs rax, 9223372036854775804 and rsi, rax - cmp rsi, 9 - jae .LBB5_3 - mov edx, 1 - xor eax, eax + cmp rsi, 8 + ja .LBB5_3 +.LBB5_2: + xor edx, edx + test dil, 3 + sete dl + xor edi, edi + mov rax, rdi ret .LBB5_3: add rsi, -9 diff --git a/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64.mca b/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64.mca index 62ea4babaf..499c58d94b 100644 --- a/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/ref_from_prefix_dynamic_padding.x86-64.mca @@ -1,6 +1,6 @@ Iterations: 100 Instructions: 1900 -Total Cycles: 608 +Total Cycles: 607 Total uOps: 2000 Dispatch Width: 4 @@ -18,17 +18,17 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 0 0.25 xor edx, edx - 1 1 0.33 mov eax, 0 1 1 0.33 test dil, 3 - 1 1 1.00 je .LBB5_1 - 1 1 1.00 U ret + 1 1 1.00 jne .LBB5_2 1 1 0.33 movabs rax, 9223372036854775804 1 1 0.33 and rsi, rax - 1 1 0.33 cmp rsi, 9 - 1 1 1.00 jae .LBB5_3 - 1 1 0.33 mov edx, 1 - 1 0 0.25 xor eax, eax + 1 1 0.33 cmp rsi, 8 + 1 1 1.00 ja .LBB5_3 + 1 0 0.25 xor edx, edx + 1 1 0.33 test dil, 3 + 1 1 0.50 sete dl + 1 0 0.25 xor edi, edi + 1 1 0.33 mov rax, rdi 1 1 1.00 U ret 1 1 0.33 add rsi, -9 1 1 0.33 movabs rcx, -6148914691236517205 @@ -52,26 +52,26 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.00 6.00 - 6.00 - - + - - 6.00 5.99 - 6.01 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: + - - 0.01 0.98 - 0.01 - - test dil, 3 + - - - - - 1.00 - - jne .LBB5_2 + - - 0.98 0.01 - 0.01 - - movabs rax, 9223372036854775804 + - - 0.01 0.99 - - - - and rsi, rax + - - 0.01 0.99 - - - - cmp rsi, 8 + - - - - - 1.00 - - ja .LBB5_3 - - - - - - - - xor edx, edx - - - 0.01 0.98 - 0.01 - - mov eax, 0 - - - 0.98 0.01 - 0.01 - - test dil, 3 - - - - - - 1.00 - - je .LBB5_1 - - - - - - 1.00 - - ret - - - 0.01 0.99 - - - - movabs rax, 9223372036854775804 - - - - 1.00 - - - - and rsi, rax - - - - 1.00 - - - - cmp rsi, 9 - - - - - - 1.00 - - jae .LBB5_3 - - - 1.00 - - - - - mov edx, 1 - - - - - - - - - xor eax, eax + - - 0.99 0.01 - - - - test dil, 3 + - - 0.99 - - 0.01 - - sete dl + - - - - - - - - xor edi, edi + - - - 0.01 - 0.99 - - mov rax, rdi - - - - - 1.00 - - ret - - - 0.02 0.02 - 0.96 - - add rsi, -9 - - - 0.99 0.01 - - - - movabs rcx, -6148914691236517205 - - - 0.01 0.99 - - - - mov rax, rsi + - - 0.01 0.99 - - - - add rsi, -9 + - - - 1.00 - - - - movabs rcx, -6148914691236517205 + - - 1.00 - - - - - mov rax, rsi - - 1.00 1.00 - - - - mul rcx - - 1.00 - - - - - shr rdx - - - 0.98 - - 0.02 - - mov rax, rdi + - - - 0.01 - 0.99 - - mov rax, rdi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64 b/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64 index fe6332c910..3f2e874678 100644 --- a/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64 +++ b/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64 @@ -1,17 +1,10 @@ bench_ref_from_prefix_dynamic_size: - xor edx, edx - mov eax, 0 - test dil, 1 - jne .LBB5_4 - cmp rsi, 4 - jae .LBB5_3 - mov edx, 1 - xor eax, eax - ret -.LBB5_3: - add rsi, -4 - shr rsi mov rdx, rsi - mov rax, rdi -.LBB5_4: + xor eax, eax + sub rdx, 4 + mov rcx, rdi + cmovb rcx, rax + shr rdx + test dil, 1 + cmove rax, rcx ret diff --git a/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64.mca b/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64.mca index 3900a59461..7ee51e0b8a 100644 --- a/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64.mca +++ b/zerocopy/benches/ref_from_prefix_dynamic_size.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1400 -Total Cycles: 405 -Total uOps: 1400 +Instructions: 900 +Total Cycles: 339 +Total uOps: 1100 Dispatch Width: 4 -uOps Per Cycle: 3.46 -IPC: 3.46 -Block RThroughput: 4.0 +uOps Per Cycle: 3.24 +IPC: 2.65 +Block RThroughput: 2.8 Instruction Info: @@ -18,19 +18,14 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 0 0.25 xor edx, edx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test dil, 1 - 1 1 1.00 jne .LBB5_4 - 1 1 0.33 cmp rsi, 4 - 1 1 1.00 jae .LBB5_3 - 1 1 0.33 mov edx, 1 - 1 0 0.25 xor eax, eax - 1 1 1.00 U ret - 1 1 0.33 add rsi, -4 - 1 1 0.50 shr rsi 1 1 0.33 mov rdx, rsi - 1 1 0.33 mov rax, rdi + 1 0 0.25 xor eax, eax + 1 1 0.33 sub rdx, 4 + 1 1 0.33 mov rcx, rdi + 2 2 0.67 cmovb rcx, rax + 1 1 0.50 shr rdx + 1 1 0.33 test dil, 1 + 2 2 0.67 cmove rax, rcx 1 1 1.00 U ret @@ -47,21 +42,16 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 3.99 3.99 - 4.02 - - + - - 3.33 3.33 - 3.34 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - - - - - - - xor edx, edx - - - 0.01 0.98 - 0.01 - - mov eax, 0 - - - 0.98 0.02 - - - - test dil, 1 - - - - - - 1.00 - - jne .LBB5_4 - - - 0.02 0.98 - - - - cmp rsi, 4 - - - - - - 1.00 - - jae .LBB5_3 - - - 0.98 0.01 - 0.01 - - mov edx, 1 + - - 0.33 0.66 - 0.01 - - mov rdx, rsi - - - - - - - - xor eax, eax - - - - - - 1.00 - - ret - - - 0.01 0.99 - - - - add rsi, -4 - - - 1.00 - - - - - shr rsi - - - - 1.00 - - - - mov rdx, rsi - - - 0.99 0.01 - - - - mov rax, rdi + - - 0.01 - - 0.99 - - sub rdx, 4 + - - 0.66 0.34 - - - - mov rcx, rdi + - - 1.00 1.00 - - - - cmovb rcx, rax + - - 0.33 - - 0.67 - - shr rdx + - - - 0.33 - 0.67 - - test dil, 1 + - - 1.00 1.00 - - - - cmove rax, rcx - - - - - 1.00 - - ret diff --git a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64 b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64 index c3d10b5fc6..7d11849dd5 100644 --- a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64 +++ b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64 @@ -1,27 +1,21 @@ bench_ref_from_suffix_with_elems_dynamic_padding: - movabs rax, 3074457345618258598 - cmp rdx, rax - ja .LBB5_1 - lea r8d, [rsi + rdi] - xor ecx, ecx - mov eax, 0 - test r8b, 3 - je .LBB5_3 - mov rdx, rcx - ret + movabs rcx, 3074457345618258598 + cmp rdx, rcx + ja .LBB5_3 + mov rax, rdi + lea rcx, [rdx + 2*rdx] + or rcx, 3 + add rcx, 9 + add edi, esi + test dil, 3 + setne dil + sub rsi, rcx + setb cl + or cl, dil + je .LBB5_4 .LBB5_3: - lea rax, [rdx + 2*rdx] - or rax, 3 - add rax, 9 - sub rsi, rax - jae .LBB5_4 -.LBB5_1: xor eax, eax - mov edx, 1 ret .LBB5_4: - add rdi, rsi - mov rcx, rdx - mov rax, rdi - mov rdx, rcx + add rax, rsi ret diff --git a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64.mca b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64.mca index 92e6280bb4..d7850d93bc 100644 --- a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2300 -Total Cycles: 706 -Total uOps: 2300 +Instructions: 1800 +Total Cycles: 572 +Total uOps: 1800 Dispatch Width: 4 -uOps Per Cycle: 3.26 -IPC: 3.26 -Block RThroughput: 6.0 +uOps Per Cycle: 3.15 +IPC: 3.15 +Block RThroughput: 4.5 Instruction Info: @@ -18,28 +18,23 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 1 0.33 movabs rax, 3074457345618258598 - 1 1 0.33 cmp rdx, rax - 1 1 1.00 ja .LBB5_1 - 1 1 0.50 lea r8d, [rsi + rdi] - 1 0 0.25 xor ecx, ecx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test r8b, 3 - 1 1 1.00 je .LBB5_3 - 1 1 0.33 mov rdx, rcx - 1 1 1.00 U ret - 1 1 0.50 lea rax, [rdx + 2*rdx] - 1 1 0.33 or rax, 3 - 1 1 0.33 add rax, 9 - 1 1 0.33 sub rsi, rax - 1 1 1.00 jae .LBB5_4 + 1 1 0.33 movabs rcx, 3074457345618258598 + 1 1 0.33 cmp rdx, rcx + 1 1 1.00 ja .LBB5_3 + 1 1 0.33 mov rax, rdi + 1 1 0.50 lea rcx, [rdx + 2*rdx] + 1 1 0.33 or rcx, 3 + 1 1 0.33 add rcx, 9 + 1 1 0.33 add edi, esi + 1 1 0.33 test dil, 3 + 1 1 0.50 setne dil + 1 1 0.33 sub rsi, rcx + 1 1 0.50 setb cl + 1 1 0.33 or cl, dil + 1 1 1.00 je .LBB5_4 1 0 0.25 xor eax, eax - 1 1 0.33 mov edx, 1 1 1 1.00 U ret - 1 1 0.33 add rdi, rsi - 1 1 0.33 mov rcx, rdx - 1 1 0.33 mov rax, rdi - 1 1 0.33 mov rdx, rcx + 1 1 0.33 add rax, rsi 1 1 1.00 U ret @@ -56,30 +51,25 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.99 7.00 - 7.01 - - + - - 5.66 5.66 - 5.68 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - - 0.99 - 0.01 - - movabs rax, 3074457345618258598 - - - 0.01 0.50 - 0.49 - - cmp rdx, rax - - - - - - 1.00 - - ja .LBB5_1 - - - - 1.00 - - - - lea r8d, [rsi + rdi] - - - - - - - - - xor ecx, ecx - - - 0.50 0.49 - 0.01 - - mov eax, 0 - - - 0.49 0.51 - - - - test r8b, 3 - - - - - - 1.00 - - je .LBB5_3 - - - 0.51 0.49 - - - - mov rdx, rcx - - - - - - 1.00 - - ret - - - 0.50 0.50 - - - - lea rax, [rdx + 2*rdx] - - - 1.00 - - - - - or rax, 3 - - - 1.00 - - - - - add rax, 9 - - - 0.99 0.01 - - - - sub rsi, rax - - - - - - 1.00 - - jae .LBB5_4 + - - 0.67 - - 0.33 - - movabs rcx, 3074457345618258598 + - - 0.34 - - 0.66 - - cmp rdx, rcx + - - - - - 1.00 - - ja .LBB5_3 + - - 0.33 0.01 - 0.66 - - mov rax, rdi + - - 0.33 0.67 - - - - lea rcx, [rdx + 2*rdx] + - - 0.01 0.99 - - - - or rcx, 3 + - - 0.01 0.99 - - - - add rcx, 9 + - - 0.99 - - 0.01 - - add edi, esi + - - 0.99 0.01 - - - - test dil, 3 + - - 0.99 - - 0.01 - - setne dil + - - - 1.00 - - - - sub rsi, rcx + - - 1.00 - - - - - setb cl + - - - 0.99 - 0.01 - - or cl, dil + - - - - - 1.00 - - je .LBB5_4 - - - - - - - - xor eax, eax - - - - 1.00 - - - - mov edx, 1 - - - - - 1.00 - - ret - - - 1.00 - - - - - add rdi, rsi - - - - 1.00 - - - - mov rcx, rdx - - - 0.99 0.01 - - - - mov rax, rdi - - - - 0.50 - 0.50 - - mov rdx, rcx + - - - 1.00 - - - - add rax, rsi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64 b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64 index bdca571924..7ee0278dbf 100644 --- a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64 +++ b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64 @@ -1,23 +1,18 @@ bench_ref_from_suffix_with_elems_dynamic_size: - movabs rax, 4611686018427387901 - cmp rdx, rax - ja .LBB5_1 - lea r8d, [rsi + rdi] - xor ecx, ecx - mov eax, 0 - test r8b, 1 - jne .LBB5_5 - lea rax, [2*rdx + 4] - sub rsi, rax - jae .LBB5_4 -.LBB5_1: + movabs rcx, 4611686018427387901 + cmp rdx, rcx + ja .LBB5_3 + mov rax, rdi + lea rcx, [2*rdx + 4] + add edi, esi + sub rsi, rcx + setb cl + or cl, dil + test cl, 1 + je .LBB5_4 +.LBB5_3: xor eax, eax - mov edx, 1 ret .LBB5_4: - add rdi, rsi - mov rcx, rdx - mov rax, rdi -.LBB5_5: - mov rdx, rcx + add rax, rsi ret diff --git a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64.mca b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64.mca index 6d9de0b3eb..882316de3d 100644 --- a/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64.mca +++ b/zerocopy/benches/ref_from_suffix_with_elems_dynamic_size.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1900 -Total Cycles: 571 -Total uOps: 1900 +Instructions: 1500 +Total Cycles: 472 +Total uOps: 1500 Dispatch Width: 4 -uOps Per Cycle: 3.33 -IPC: 3.33 -Block RThroughput: 5.0 +uOps Per Cycle: 3.18 +IPC: 3.18 +Block RThroughput: 4.0 Instruction Info: @@ -18,24 +18,20 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 1 0.33 movabs rax, 4611686018427387901 - 1 1 0.33 cmp rdx, rax - 1 1 1.00 ja .LBB5_1 - 1 1 0.50 lea r8d, [rsi + rdi] - 1 0 0.25 xor ecx, ecx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test r8b, 1 - 1 1 1.00 jne .LBB5_5 - 1 1 0.50 lea rax, [2*rdx + 4] - 1 1 0.33 sub rsi, rax - 1 1 1.00 jae .LBB5_4 + 1 1 0.33 movabs rcx, 4611686018427387901 + 1 1 0.33 cmp rdx, rcx + 1 1 1.00 ja .LBB5_3 + 1 1 0.33 mov rax, rdi + 1 1 0.50 lea rcx, [2*rdx + 4] + 1 1 0.33 add edi, esi + 1 1 0.33 sub rsi, rcx + 1 1 0.50 setb cl + 1 1 0.33 or cl, dil + 1 1 0.33 test cl, 1 + 1 1 1.00 je .LBB5_4 1 0 0.25 xor eax, eax - 1 1 0.33 mov edx, 1 1 1 1.00 U ret - 1 1 0.33 add rdi, rsi - 1 1 0.33 mov rcx, rdx - 1 1 0.33 mov rax, rdi - 1 1 0.33 mov rdx, rcx + 1 1 0.33 add rax, rsi 1 1 1.00 U ret @@ -52,26 +48,22 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 5.66 5.66 - 5.68 - - + - - 4.66 4.66 - 4.68 - - Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.66 0.33 - 0.01 - - movabs rax, 4611686018427387901 - - - 0.01 0.99 - - - - cmp rdx, rax - - - - - - 1.00 - - ja .LBB5_1 - - - 0.99 0.01 - - - - lea r8d, [rsi + rdi] - - - - - - - - - xor ecx, ecx - - - 0.33 0.33 - 0.34 - - mov eax, 0 - - - 0.33 0.34 - 0.33 - - test r8b, 1 - - - - - - 1.00 - - jne .LBB5_5 - - - 0.34 0.66 - - - - lea rax, [2*rdx + 4] - - - - 1.00 - - - - sub rsi, rax - - - - - - 1.00 - - jae .LBB5_4 + - - 0.99 - - 0.01 - - movabs rcx, 4611686018427387901 + - - 1.00 - - - - - cmp rdx, rcx + - - - - - 1.00 - - ja .LBB5_3 + - - 0.63 0.02 - 0.35 - - mov rax, rdi + - - 0.35 0.65 - - - - lea rcx, [2*rdx + 4] + - - 0.65 0.03 - 0.32 - - add edi, esi + - - 0.03 0.97 - - - - sub rsi, rcx + - - 1.00 - - - - - setb cl + - - - 1.00 - - - - or cl, dil + - - 0.01 0.99 - - - - test cl, 1 + - - - - - 1.00 - - je .LBB5_4 - - - - - - - - xor eax, eax - - - 1.00 - - - - - mov edx, 1 - - - - - 1.00 - - ret - - - - 1.00 - - - - add rdi, rsi - - - 1.00 - - - - - mov rcx, rdx - - - 0.32 0.68 - - - - mov rax, rdi - - - 0.68 0.32 - - - - mov rdx, rcx + - - - 1.00 - - - - add rax, rsi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_read_from_bytes.x86-64 b/zerocopy/benches/try_read_from_bytes.x86-64 index 08088a08fd..2135facc26 100644 --- a/zerocopy/benches/try_read_from_bytes.x86-64 +++ b/zerocopy/benches/try_read_from_bytes.x86-64 @@ -1,23 +1,11 @@ bench_try_read_from_bytes_static_size: - mov ax, -16191 + mov eax, 49345 cmp rsi, 6 - jne .LBB5_1 - mov ecx, dword ptr [rdi] - movzx edx, cx - cmp edx, 49344 - jne .LBB5_4 - movzx eax, word ptr [rdi + 4] - shl rax, 32 - or rcx, rax - shr rcx, 16 - mov ax, -16192 -.LBB5_4: - shl rcx, 16 - movzx eax, ax - or rax, rcx - ret -.LBB5_1: - shl rcx, 16 - movzx eax, ax - or rax, rcx + jne .LBB5_3 + cmp word ptr [rdi], -16192 + jne .LBB5_3 + mov eax, dword ptr [rdi + 2] + shl rax, 16 + or rax, 49344 +.LBB5_3: ret diff --git a/zerocopy/benches/try_read_from_bytes.x86-64.mca b/zerocopy/benches/try_read_from_bytes.x86-64.mca index 385e6a4802..f5568fef5a 100644 --- a/zerocopy/benches/try_read_from_bytes.x86-64.mca +++ b/zerocopy/benches/try_read_from_bytes.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2000 -Total Cycles: 608 -Total uOps: 2000 +Instructions: 900 +Total Cycles: 305 +Total uOps: 1000 Dispatch Width: 4 -uOps Per Cycle: 3.29 -IPC: 3.29 -Block RThroughput: 5.0 +uOps Per Cycle: 3.28 +IPC: 2.95 +Block RThroughput: 3.0 Instruction Info: @@ -18,25 +18,14 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 1 0.33 mov ax, -16191 + 1 1 0.33 mov eax, 49345 1 1 0.33 cmp rsi, 6 - 1 1 1.00 jne .LBB5_1 - 1 5 0.50 * mov ecx, dword ptr [rdi] - 1 1 0.33 movzx edx, cx - 1 1 0.33 cmp edx, 49344 - 1 1 1.00 jne .LBB5_4 - 1 5 0.50 * movzx eax, word ptr [rdi + 4] - 1 1 0.50 shl rax, 32 - 1 1 0.33 or rcx, rax - 1 1 0.50 shr rcx, 16 - 1 1 0.33 mov ax, -16192 - 1 1 0.50 shl rcx, 16 - 1 1 0.33 movzx eax, ax - 1 1 0.33 or rax, rcx - 1 1 1.00 U ret - 1 1 0.50 shl rcx, 16 - 1 1 0.33 movzx eax, ax - 1 1 0.33 or rax, rcx + 1 1 1.00 jne .LBB5_3 + 2 6 0.50 * cmp word ptr [rdi], -16192 + 1 1 1.00 jne .LBB5_3 + 1 5 0.50 * mov eax, dword ptr [rdi + 2] + 1 1 0.50 shl rax, 16 + 1 1 0.33 or rax, 49344 1 1 1.00 U ret @@ -53,27 +42,16 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 5.99 5.99 - 6.02 1.00 1.00 + - - 2.50 2.49 - 3.01 1.00 1.00 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - - 0.99 - 0.01 - - mov ax, -16191 - - - - 0.01 - 0.99 - - cmp rsi, 6 - - - - - - 1.00 - - jne .LBB5_1 - - - - - - - - 1.00 mov ecx, dword ptr [rdi] - - - 0.98 - - 0.02 - - movzx edx, cx - - - 0.99 0.01 - - - - cmp edx, 49344 - - - - - - 1.00 - - jne .LBB5_4 - - - - - - - 1.00 - movzx eax, word ptr [rdi + 4] - - - 0.01 - - 0.99 - - shl rax, 32 - - - 0.02 0.98 - - - - or rcx, rax - - - 1.00 - - - - - shr rcx, 16 - - - 0.99 0.01 - - - - mov ax, -16192 - - - 1.00 - - - - - shl rcx, 16 - - - - 1.00 - - - - movzx eax, ax - - - - 1.00 - - - - or rax, rcx - - - - - - 1.00 - - ret - - - 1.00 - - - - - shl rcx, 16 - - - - 1.00 - - - - movzx eax, ax - - - - 0.99 - 0.01 - - or rax, rcx + - - 0.50 0.49 - 0.01 - - mov eax, 49345 + - - 0.49 0.51 - - - - cmp rsi, 6 + - - - - - 1.00 - - jne .LBB5_3 + - - 0.47 0.53 - - 0.43 0.57 cmp word ptr [rdi], -16192 + - - - - - 1.00 - - jne .LBB5_3 + - - - - - - 0.57 0.43 mov eax, dword ptr [rdi + 2] + - - 1.00 - - - - - shl rax, 16 + - - 0.04 0.96 - - - - or rax, 49344 - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_read_from_prefix.x86-64 b/zerocopy/benches/try_read_from_prefix.x86-64 index d3e1edc3ea..0f820efff5 100644 --- a/zerocopy/benches/try_read_from_prefix.x86-64 +++ b/zerocopy/benches/try_read_from_prefix.x86-64 @@ -1,16 +1,11 @@ bench_try_read_from_prefix_static_size: mov eax, 49345 cmp rsi, 6 - jb .LBB5_2 - mov eax, dword ptr [rdi] - movzx ecx, word ptr [rdi + 4] - shl rcx, 32 - or rcx, rax - movzx eax, cx - and rcx, -65536 - or rcx, 49344 - cmp eax, 49344 - mov eax, 49345 - cmove rax, rcx -.LBB5_2: + jb .LBB5_3 + cmp word ptr [rdi], -16192 + jne .LBB5_3 + mov eax, dword ptr [rdi + 2] + shl rax, 16 + or rax, 49344 +.LBB5_3: ret diff --git a/zerocopy/benches/try_read_from_prefix.x86-64.mca b/zerocopy/benches/try_read_from_prefix.x86-64.mca index 40401d89e8..e0ba4c8320 100644 --- a/zerocopy/benches/try_read_from_prefix.x86-64.mca +++ b/zerocopy/benches/try_read_from_prefix.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1400 -Total Cycles: 442 -Total uOps: 1500 +Instructions: 900 +Total Cycles: 305 +Total uOps: 1000 Dispatch Width: 4 -uOps Per Cycle: 3.39 -IPC: 3.17 -Block RThroughput: 3.8 +uOps Per Cycle: 3.28 +IPC: 2.95 +Block RThroughput: 3.0 Instruction Info: @@ -20,17 +20,12 @@ Instruction Info: [1] [2] [3] [4] [5] [6] Instructions: 1 1 0.33 mov eax, 49345 1 1 0.33 cmp rsi, 6 - 1 1 1.00 jb .LBB5_2 - 1 5 0.50 * mov eax, dword ptr [rdi] - 1 5 0.50 * movzx ecx, word ptr [rdi + 4] - 1 1 0.50 shl rcx, 32 - 1 1 0.33 or rcx, rax - 1 1 0.33 movzx eax, cx - 1 1 0.33 and rcx, -65536 - 1 1 0.33 or rcx, 49344 - 1 1 0.33 cmp eax, 49344 - 1 1 0.33 mov eax, 49345 - 2 2 0.67 cmove rax, rcx + 1 1 1.00 jb .LBB5_3 + 2 6 0.50 * cmp word ptr [rdi], -16192 + 1 1 1.00 jne .LBB5_3 + 1 5 0.50 * mov eax, dword ptr [rdi + 2] + 1 1 0.50 shl rax, 16 + 1 1 0.33 or rax, 49344 1 1 1.00 U ret @@ -47,21 +42,16 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 4.33 4.33 - 4.34 1.00 1.00 + - - 2.50 2.49 - 3.01 1.00 1.00 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.65 0.01 - 0.34 - - mov eax, 49345 - - - 0.01 0.33 - 0.66 - - cmp rsi, 6 - - - - - - 1.00 - - jb .LBB5_2 - - - - - - - - 1.00 mov eax, dword ptr [rdi] - - - - - - - 1.00 - movzx ecx, word ptr [rdi + 4] - - - 0.65 - - 0.35 - - shl rcx, 32 - - - - 0.67 - 0.33 - - or rcx, rax - - - 0.01 0.99 - - - - movzx eax, cx - - - 0.99 0.01 - - - - and rcx, -65536 - - - 0.01 0.99 - - - - or rcx, 49344 - - - 0.99 0.01 - - - - cmp eax, 49344 - - - 0.02 0.33 - 0.65 - - mov eax, 49345 - - - 1.00 0.99 - 0.01 - - cmove rax, rcx + - - 0.50 0.49 - 0.01 - - mov eax, 49345 + - - 0.49 0.51 - - - - cmp rsi, 6 + - - - - - 1.00 - - jb .LBB5_3 + - - 0.47 0.53 - - 0.43 0.57 cmp word ptr [rdi], -16192 + - - - - - 1.00 - - jne .LBB5_3 + - - - - - - 0.57 0.43 mov eax, dword ptr [rdi + 2] + - - 1.00 - - - - - shl rax, 16 + - - 0.04 0.96 - - - - or rax, 49344 - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_read_from_suffix.x86-64 b/zerocopy/benches/try_read_from_suffix.x86-64 index 095e326f04..a91c074e18 100644 --- a/zerocopy/benches/try_read_from_suffix.x86-64 +++ b/zerocopy/benches/try_read_from_suffix.x86-64 @@ -1,18 +1,11 @@ bench_try_read_from_suffix_static_size: mov eax, 49345 cmp rsi, 6 - jb .LBB5_2 - mov eax, dword ptr [rdi + rsi - 6] - movzx ecx, word ptr [rdi + rsi - 2] - shl rcx, 32 - or rcx, rax - movzx edx, cx - xor eax, eax - cmp edx, 49344 - cmovne rcx, rsi - sete al - and rcx, -65536 - xor rax, 49345 - or rax, rcx -.LBB5_2: + jb .LBB5_3 + cmp word ptr [rdi + rsi - 6], -16192 + jne .LBB5_3 + mov eax, dword ptr [rdi + rsi - 4] + shl rax, 16 + or rax, 49344 +.LBB5_3: ret diff --git a/zerocopy/benches/try_read_from_suffix.x86-64.mca b/zerocopy/benches/try_read_from_suffix.x86-64.mca index d3eaadbb8a..e42f6b93ef 100644 --- a/zerocopy/benches/try_read_from_suffix.x86-64.mca +++ b/zerocopy/benches/try_read_from_suffix.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1600 -Total Cycles: 478 -Total uOps: 1700 +Instructions: 900 +Total Cycles: 305 +Total uOps: 1000 Dispatch Width: 4 -uOps Per Cycle: 3.56 -IPC: 3.35 -Block RThroughput: 4.3 +uOps Per Cycle: 3.28 +IPC: 2.95 +Block RThroughput: 3.0 Instruction Info: @@ -20,19 +20,12 @@ Instruction Info: [1] [2] [3] [4] [5] [6] Instructions: 1 1 0.33 mov eax, 49345 1 1 0.33 cmp rsi, 6 - 1 1 1.00 jb .LBB5_2 - 1 5 0.50 * mov eax, dword ptr [rdi + rsi - 6] - 1 5 0.50 * movzx ecx, word ptr [rdi + rsi - 2] - 1 1 0.50 shl rcx, 32 - 1 1 0.33 or rcx, rax - 1 1 0.33 movzx edx, cx - 1 0 0.25 xor eax, eax - 1 1 0.33 cmp edx, 49344 - 2 2 0.67 cmovne rcx, rsi - 1 1 0.50 sete al - 1 1 0.33 and rcx, -65536 - 1 1 0.33 xor rax, 49345 - 1 1 0.33 or rax, rcx + 1 1 1.00 jb .LBB5_3 + 2 6 0.50 * cmp word ptr [rdi + rsi - 6], -16192 + 1 1 1.00 jne .LBB5_3 + 1 5 0.50 * mov eax, dword ptr [rdi + rsi - 4] + 1 1 0.50 shl rax, 16 + 1 1 0.33 or rax, 49344 1 1 1.00 U ret @@ -49,23 +42,16 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 4.66 4.66 - 4.68 1.00 1.00 + - - 2.50 2.49 - 3.01 1.00 1.00 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.32 0.01 - 0.67 - - mov eax, 49345 - - - 0.62 0.02 - 0.36 - - cmp rsi, 6 - - - - - - 1.00 - - jb .LBB5_2 - - - - - - - - 1.00 mov eax, dword ptr [rdi + rsi - 6] - - - - - - - 1.00 - movzx ecx, word ptr [rdi + rsi - 2] - - - 0.37 - - 0.63 - - shl rcx, 32 - - - 0.99 0.01 - - - - or rcx, rax - - - 1.00 - - - - - movzx edx, cx - - - - - - - - - xor eax, eax - - - 0.35 0.64 - 0.01 - - cmp edx, 49344 - - - 1.00 1.00 - - - - cmovne rcx, rsi - - - - - - 1.00 - - sete al - - - 0.01 0.99 - - - - and rcx, -65536 - - - - 1.00 - - - - xor rax, 49345 - - - - 0.99 - 0.01 - - or rax, rcx + - - 0.50 0.49 - 0.01 - - mov eax, 49345 + - - 0.49 0.51 - - - - cmp rsi, 6 + - - - - - 1.00 - - jb .LBB5_3 + - - 0.47 0.53 - - 0.43 0.57 cmp word ptr [rdi + rsi - 6], -16192 + - - - - - 1.00 - - jne .LBB5_3 + - - - - - - 0.57 0.43 mov eax, dword ptr [rdi + rsi - 4] + - - 1.00 - - - - - shl rax, 16 + - - 0.04 0.96 - - - - or rax, 49344 - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64 b/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64 index d832cb7ecf..aa4ddaaca7 100644 --- a/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64 +++ b/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64 @@ -1,15 +1,14 @@ bench_try_ref_from_prefix_dynamic_padding: - xor edx, edx - mov eax, 0 test dil, 3 - je .LBB5_1 - ret -.LBB5_1: + jne .LBB5_2 movabs rax, 9223372036854775804 and rsi, rax - cmp rsi, 9 - jae .LBB5_3 - mov edx, 1 + cmp rsi, 8 + ja .LBB5_3 +.LBB5_2: + xor edx, edx + test dil, 3 + sete dl xor eax, eax ret .LBB5_3: diff --git a/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64.mca b/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64.mca index 482112a39b..f595d499cf 100644 --- a/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/try_ref_from_prefix_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2600 -Total Cycles: 843 -Total uOps: 2900 +Instructions: 2500 +Total Cycles: 808 +Total uOps: 2800 Dispatch Width: 4 -uOps Per Cycle: 3.44 -IPC: 3.08 -Block RThroughput: 7.3 +uOps Per Cycle: 3.47 +IPC: 3.09 +Block RThroughput: 7.0 Instruction Info: @@ -18,16 +18,15 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 0 0.25 xor edx, edx - 1 1 0.33 mov eax, 0 1 1 0.33 test dil, 3 - 1 1 1.00 je .LBB5_1 - 1 1 1.00 U ret + 1 1 1.00 jne .LBB5_2 1 1 0.33 movabs rax, 9223372036854775804 1 1 0.33 and rsi, rax - 1 1 0.33 cmp rsi, 9 - 1 1 1.00 jae .LBB5_3 - 1 1 0.33 mov edx, 1 + 1 1 0.33 cmp rsi, 8 + 1 1 1.00 ja .LBB5_3 + 1 0 0.25 xor edx, edx + 1 1 0.33 test dil, 3 + 1 1 0.50 sete dl 1 0 0.25 xor eax, eax 1 1 1.00 U ret 1 1 0.33 add rsi, -9 @@ -59,33 +58,32 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 8.33 8.33 - 8.34 0.50 0.50 + - - 8.00 8.00 - 8.00 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: + - - 0.97 0.02 - 0.01 - - test dil, 3 + - - - - - 1.00 - - jne .LBB5_2 + - - 0.99 0.01 - - - - movabs rax, 9223372036854775804 + - - 0.02 0.98 - - - - and rsi, rax + - - 0.03 0.97 - - - - cmp rsi, 8 + - - - - - 1.00 - - ja .LBB5_3 - - - - - - - - xor edx, edx - - - 0.32 0.34 - 0.34 - - mov eax, 0 - - - 0.34 0.33 - 0.33 - - test dil, 3 - - - - - - 1.00 - - je .LBB5_1 - - - - - - 1.00 - - ret - - - 0.35 0.65 - - - - movabs rax, 9223372036854775804 - - - 0.96 0.03 - 0.01 - - and rsi, rax - - - 0.01 0.97 - 0.02 - - cmp rsi, 9 - - - - - - 1.00 - - jae .LBB5_3 - - - 0.67 0.01 - 0.32 - - mov edx, 1 + - - 0.98 0.02 - - - - test dil, 3 + - - 0.99 - - 0.01 - - sete dl - - - - - - - - xor eax, eax - - - - - 1.00 - - ret - - - 0.02 0.34 - 0.64 - - add rsi, -9 - - - 0.33 0.66 - 0.01 - - movabs rcx, -6148914691236517205 - - - 0.66 0.34 - - - - mov rax, rsi + - - 0.98 0.02 - - - - add rsi, -9 + - - 0.02 0.98 - - - - movabs rcx, -6148914691236517205 + - - 0.98 0.02 - - - - mov rax, rsi - - 1.00 1.00 - - - - mul rcx - - - 0.01 0.99 - - - - mov rax, rdx - - - 0.99 - - 0.01 - - shr rax + - - - 0.01 - 0.99 - - mov rax, rdx + - - 0.01 - - 0.99 - - shr rax - - - - - - 0.50 0.50 movzx ecx, word ptr [rdi] - - - 0.33 0.03 - 0.64 - - cmp cx, -16192 - - - 0.01 0.31 - 0.68 - - mov edx, 2 - - - 1.00 1.00 - - - - cmove rdx, rax + - - - 0.99 - 0.01 - - cmp cx, -16192 + - - 0.99 0.01 - - - - mov edx, 2 + - - 0.01 1.00 - 0.99 - - cmove rdx, rax - - - - - - - - xor eax, eax - - - 0.33 0.33 - 0.34 - - cmp ecx, 49344 - - - 1.00 1.00 - - - - cmove rax, rdi + - - 0.02 0.98 - - - - cmp ecx, 49344 + - - 0.01 0.99 - 1.00 - - cmove rax, rdi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64 b/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64 index be7f34b9f8..7cd0a48b21 100644 --- a/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64 +++ b/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64 @@ -1,14 +1,15 @@ bench_try_ref_from_prefix_dynamic_size: - xor edx, edx - mov eax, 0 - test dil, 1 - jne .LBB5_4 cmp rsi, 4 - jae .LBB5_3 - mov edx, 1 + setb al + or al, dil + test al, 1 + je .LBB5_2 + and edi, 1 + xor rdi, 1 xor eax, eax + mov rdx, rdi ret -.LBB5_3: +.LBB5_2: add rsi, -4 shr rsi movzx ecx, word ptr [rdi] @@ -18,5 +19,4 @@ bench_try_ref_from_prefix_dynamic_size: xor eax, eax cmp cx, -16192 cmove rax, rdi -.LBB5_4: ret diff --git a/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64.mca b/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64.mca index 11706defe1..43fdb855ee 100644 --- a/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64.mca +++ b/zerocopy/benches/try_ref_from_prefix_dynamic_size.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 1900 -Total Cycles: 573 -Total uOps: 2100 +Instructions: 2000 +Total Cycles: 643 +Total uOps: 2200 Dispatch Width: 4 -uOps Per Cycle: 3.66 -IPC: 3.32 -Block RThroughput: 5.3 +uOps Per Cycle: 3.42 +IPC: 3.11 +Block RThroughput: 5.5 Instruction Info: @@ -18,14 +18,15 @@ Instruction Info: [6]: HasSideEffects (U) [1] [2] [3] [4] [5] [6] Instructions: - 1 0 0.25 xor edx, edx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test dil, 1 - 1 1 1.00 jne .LBB5_4 1 1 0.33 cmp rsi, 4 - 1 1 1.00 jae .LBB5_3 - 1 1 0.33 mov edx, 1 + 1 1 0.50 setb al + 1 1 0.33 or al, dil + 1 1 0.33 test al, 1 + 1 1 1.00 je .LBB5_2 + 1 1 0.33 and edi, 1 + 1 1 0.33 xor rdi, 1 1 0 0.25 xor eax, eax + 1 1 0.33 mov rdx, rdi 1 1 1.00 U ret 1 1 0.33 add rsi, -4 1 1 0.50 shr rsi @@ -52,26 +53,27 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 5.66 5.67 - 5.67 0.50 0.50 + - - 6.33 6.33 - 6.34 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - - - - - - - xor edx, edx - - - 0.30 0.37 - 0.33 - - mov eax, 0 - - - 0.35 0.32 - 0.33 - - test dil, 1 - - - - - - 1.00 - - jne .LBB5_4 - - - 0.32 0.33 - 0.35 - - cmp rsi, 4 - - - - - - 1.00 - - jae .LBB5_3 - - - 0.33 0.35 - 0.32 - - mov edx, 1 + - - 0.67 0.32 - 0.01 - - cmp rsi, 4 + - - 0.99 - - 0.01 - - setb al + - - 0.19 0.48 - 0.33 - - or al, dil + - - 0.50 0.16 - 0.34 - - test al, 1 + - - - - - 1.00 - - je .LBB5_2 + - - 0.50 0.50 - - - - and edi, 1 + - - 0.49 0.34 - 0.17 - - xor rdi, 1 - - - - - - - - xor eax, eax + - - 0.16 0.84 - - - - mov rdx, rdi - - - - - 1.00 - - ret - - - 0.34 0.64 - 0.02 - - add rsi, -4 - - - 1.00 - - - - - shr rsi + - - 0.68 0.32 - - - - add rsi, -4 + - - 0.85 - - 0.15 - - shr rsi - - - - - - 0.50 0.50 movzx ecx, word ptr [rdi] - - - 0.60 0.40 - - - - cmp ecx, 49344 - - - 0.05 0.95 - - - - mov edx, 2 - - - 1.00 1.00 - - - - cmove rdx, rsi + - - - 0.50 - 0.50 - - cmp ecx, 49344 + - - 0.63 0.37 - - - - mov edx, 2 + - - 0.17 1.00 - 0.83 - - cmove rdx, rsi - - - - - - - - xor eax, eax - - - 0.37 0.31 - 0.32 - - cmp cx, -16192 - - - 1.00 1.00 - - - - cmove rax, rdi + - - 0.50 0.50 - - - - cmp cx, -16192 + - - - 1.00 - 1.00 - - cmove rax, rdi - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64 b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64 index c7530d8b68..ca483dee63 100644 --- a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64 +++ b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64 @@ -1,32 +1,23 @@ bench_try_ref_from_suffix_with_elems_dynamic_padding: movabs rax, 3074457345618258598 cmp rdx, rax - ja .LBB5_1 - lea r8d, [rsi + rdi] - xor ecx, ecx - mov eax, 0 - test r8b, 3 - je .LBB5_3 - mov rdx, rcx - ret -.LBB5_3: + ja .LBB5_3 lea rax, [rdx + 2*rdx] or rax, 3 add rax, 9 + lea ecx, [rsi + rdi] + test cl, 3 + setne cl sub rsi, rax - jae .LBB5_4 -.LBB5_1: + setb al + or al, cl + je .LBB5_4 +.LBB5_3: xor eax, eax - mov edx, 1 ret .LBB5_4: - lea r8, [rdi + rsi] - movzx esi, word ptr [rdi + rsi] - cmp si, -16192 - mov ecx, 2 - cmove rcx, rdx + lea rcx, [rdi + rsi] xor eax, eax - cmp esi, 49344 - cmove rax, r8 - mov rdx, rcx + cmp word ptr [rdi + rsi], -16192 + cmove rax, rcx ret diff --git a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64.mca b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64.mca index be736c00c2..c37a42c160 100644 --- a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64.mca +++ b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_padding.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2800 -Total Cycles: 878 -Total uOps: 3000 +Instructions: 2000 +Total Cycles: 643 +Total uOps: 2200 Dispatch Width: 4 uOps Per Cycle: 3.42 -IPC: 3.19 -Block RThroughput: 7.5 +IPC: 3.11 +Block RThroughput: 5.5 Instruction Info: @@ -20,31 +20,23 @@ Instruction Info: [1] [2] [3] [4] [5] [6] Instructions: 1 1 0.33 movabs rax, 3074457345618258598 1 1 0.33 cmp rdx, rax - 1 1 1.00 ja .LBB5_1 - 1 1 0.50 lea r8d, [rsi + rdi] - 1 0 0.25 xor ecx, ecx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test r8b, 3 - 1 1 1.00 je .LBB5_3 - 1 1 0.33 mov rdx, rcx - 1 1 1.00 U ret + 1 1 1.00 ja .LBB5_3 1 1 0.50 lea rax, [rdx + 2*rdx] 1 1 0.33 or rax, 3 1 1 0.33 add rax, 9 + 1 1 0.50 lea ecx, [rsi + rdi] + 1 1 0.33 test cl, 3 + 1 1 0.50 setne cl 1 1 0.33 sub rsi, rax - 1 1 1.00 jae .LBB5_4 + 1 1 0.50 setb al + 1 1 0.33 or al, cl + 1 1 1.00 je .LBB5_4 1 0 0.25 xor eax, eax - 1 1 0.33 mov edx, 1 1 1 1.00 U ret - 1 1 0.50 lea r8, [rdi + rsi] - 1 5 0.50 * movzx esi, word ptr [rdi + rsi] - 1 1 0.33 cmp si, -16192 - 1 1 0.33 mov ecx, 2 - 2 2 0.67 cmove rcx, rdx + 1 1 0.50 lea rcx, [rdi + rsi] 1 0 0.25 xor eax, eax - 1 1 0.33 cmp esi, 49344 - 2 2 0.67 cmove rax, r8 - 1 1 0.33 mov rdx, rcx + 2 6 0.50 * cmp word ptr [rdi + rsi], -16192 + 2 2 0.67 cmove rax, rcx 1 1 1.00 U ret @@ -61,35 +53,27 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 8.65 8.65 - 8.70 0.50 0.50 + - - 6.32 6.33 - 6.35 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.67 0.30 - 0.03 - - movabs rax, 3074457345618258598 - - - 0.01 0.99 - - - - cmp rdx, rax - - - - - - 1.00 - - ja .LBB5_1 - - - 0.99 0.01 - - - - lea r8d, [rsi + rdi] - - - - - - - - - xor ecx, ecx - - - 0.35 0.62 - 0.03 - - mov eax, 0 - - - 0.99 0.01 - - - - test r8b, 3 - - - - - - 1.00 - - je .LBB5_3 - - - 0.68 0.30 - 0.02 - - mov rdx, rcx - - - - - - 1.00 - - ret - - - 0.07 0.93 - - - - lea rax, [rdx + 2*rdx] - - - 0.06 0.35 - 0.59 - - or rax, 3 - - - 0.02 0.07 - 0.91 - - add rax, 9 - - - 0.01 0.04 - 0.95 - - sub rsi, rax - - - - - - 1.00 - - jae .LBB5_4 + - - - 0.99 - 0.01 - - movabs rax, 3074457345618258598 + - - 0.02 0.98 - - - - cmp rdx, rax + - - - - - 1.00 - - ja .LBB5_3 + - - 0.66 0.34 - - - - lea rax, [rdx + 2*rdx] + - - 0.99 - - 0.01 - - or rax, 3 + - - 0.99 0.01 - - - - add rax, 9 + - - 0.01 0.99 - - - - lea ecx, [rsi + rdi] + - - - 0.99 - 0.01 - - test cl, 3 + - - 0.02 - - 0.98 - - setne cl + - - 0.97 0.02 - 0.01 - - sub rsi, rax + - - 0.99 - - 0.01 - - setb al + - - 0.32 0.67 - 0.01 - - or al, cl + - - - - - 1.00 - - je .LBB5_4 - - - - - - - - xor eax, eax - - - 0.92 0.01 - 0.07 - - mov edx, 1 - - - - - 1.00 - - ret - - - - 1.00 - - - - lea r8, [rdi + rsi] - - - - - - - 0.50 0.50 movzx esi, word ptr [rdi + rsi] - - - 0.01 0.99 - - - - cmp si, -16192 - - - 0.88 0.04 - 0.08 - - mov ecx, 2 - - - 1.00 0.99 - 0.01 - - cmove rcx, rdx + - - - 1.00 - - - - lea rcx, [rdi + rsi] - - - - - - - - xor eax, eax - - - 0.99 0.01 - - - - cmp esi, 49344 - - - 1.00 1.00 - - - - cmove rax, r8 - - - - 0.99 - 0.01 - - mov rdx, rcx + - - 0.35 0.32 - 0.33 0.50 0.50 cmp word ptr [rdi + rsi], -16192 + - - 1.00 0.02 - 0.98 - - cmove rax, rcx - - - - - 1.00 - - ret diff --git a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64 b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64 index 952eb12de8..748eb1166e 100644 --- a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64 +++ b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64 @@ -1,28 +1,20 @@ bench_try_ref_from_suffix_with_elems_dynamic_size: movabs rax, 4611686018427387901 cmp rdx, rax - ja .LBB5_1 - lea r8d, [rsi + rdi] - xor ecx, ecx - mov eax, 0 - test r8b, 1 - jne .LBB5_5 + ja .LBB5_3 lea rax, [2*rdx + 4] + lea ecx, [rsi + rdi] sub rsi, rax - jae .LBB5_4 -.LBB5_1: + setb al + or al, cl + test al, 1 + je .LBB5_4 +.LBB5_3: xor eax, eax - mov edx, 1 ret .LBB5_4: - lea r8, [rdi + rsi] - movzx esi, word ptr [rdi + rsi] - cmp si, -16192 - mov ecx, 2 - cmove rcx, rdx + lea rcx, [rdi + rsi] xor eax, eax - cmp esi, 49344 - cmove rax, r8 -.LBB5_5: - mov rdx, rcx + cmp word ptr [rdi + rsi], -16192 + cmove rax, rcx ret diff --git a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64.mca b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64.mca index d4f78f67a2..53ff061926 100644 --- a/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64.mca +++ b/zerocopy/benches/try_ref_from_suffix_with_elems_dynamic_size.x86-64.mca @@ -1,12 +1,12 @@ Iterations: 100 -Instructions: 2400 -Total Cycles: 1107 -Total uOps: 2600 +Instructions: 1700 +Total Cycles: 544 +Total uOps: 1900 Dispatch Width: 4 -uOps Per Cycle: 2.35 -IPC: 2.17 -Block RThroughput: 6.5 +uOps Per Cycle: 3.49 +IPC: 3.13 +Block RThroughput: 4.8 Instruction Info: @@ -20,27 +20,20 @@ Instruction Info: [1] [2] [3] [4] [5] [6] Instructions: 1 1 0.33 movabs rax, 4611686018427387901 1 1 0.33 cmp rdx, rax - 1 1 1.00 ja .LBB5_1 - 1 1 0.50 lea r8d, [rsi + rdi] - 1 0 0.25 xor ecx, ecx - 1 1 0.33 mov eax, 0 - 1 1 0.33 test r8b, 1 - 1 1 1.00 jne .LBB5_5 + 1 1 1.00 ja .LBB5_3 1 1 0.50 lea rax, [2*rdx + 4] + 1 1 0.50 lea ecx, [rsi + rdi] 1 1 0.33 sub rsi, rax - 1 1 1.00 jae .LBB5_4 + 1 1 0.50 setb al + 1 1 0.33 or al, cl + 1 1 0.33 test al, 1 + 1 1 1.00 je .LBB5_4 1 0 0.25 xor eax, eax - 1 1 0.33 mov edx, 1 1 1 1.00 U ret - 1 1 0.50 lea r8, [rdi + rsi] - 1 5 0.50 * movzx esi, word ptr [rdi + rsi] - 1 1 0.33 cmp si, -16192 - 1 1 0.33 mov ecx, 2 - 2 2 0.67 cmove rcx, rdx + 1 1 0.50 lea rcx, [rdi + rsi] 1 0 0.25 xor eax, eax - 1 1 0.33 cmp esi, 49344 - 2 2 0.67 cmove rax, r8 - 1 1 0.33 mov rdx, rcx + 2 6 0.50 * cmp word ptr [rdi + rsi], -16192 + 2 2 0.67 cmove rax, rcx 1 1 1.00 U ret @@ -57,31 +50,24 @@ Resources: Resource pressure per iteration: [0] [1] [2] [3] [4] [5] [6.0] [6.1] - - - 6.99 7.00 - 8.01 0.50 0.50 + - - 5.33 5.33 - 5.34 0.50 0.50 Resource pressure by instruction: [0] [1] [2] [3] [4] [5] [6.0] [6.1] Instructions: - - - 0.02 0.95 - 0.03 - - movabs rax, 4611686018427387901 - - - 0.93 0.04 - 0.03 - - cmp rdx, rax - - - - - - 1.00 - - ja .LBB5_1 - - - 0.96 0.04 - - - - lea r8d, [rsi + rdi] - - - - - - - - - xor ecx, ecx - - - 0.95 0.02 - 0.03 - - mov eax, 0 - - - 0.95 0.05 - - - - test r8b, 1 - - - - - - 1.00 - - jne .LBB5_5 - - - 0.06 0.94 - - - - lea rax, [2*rdx + 4] - - - 0.93 0.07 - - - - sub rsi, rax - - - - - - 1.00 - - jae .LBB5_4 + - - 0.66 0.33 - 0.01 - - movabs rax, 4611686018427387901 + - - 0.02 0.98 - - - - cmp rdx, rax + - - - - - 1.00 - - ja .LBB5_3 + - - 0.98 0.02 - - - - lea rax, [2*rdx + 4] + - - 0.66 0.34 - - - - lea ecx, [rsi + rdi] + - - 0.33 0.65 - 0.02 - - sub rsi, rax + - - 0.99 - - 0.01 - - setb al + - - 0.01 0.36 - 0.63 - - or al, cl + - - 0.01 0.96 - 0.03 - - test al, 1 + - - - - - 1.00 - - je .LBB5_4 - - - - - - - - xor eax, eax - - - 0.03 0.95 - 0.02 - - mov edx, 1 - - - - - 1.00 - - ret - - - 0.97 0.03 - - - - lea r8, [rdi + rsi] - - - - - - - 0.50 0.50 movzx esi, word ptr [rdi + rsi] - - - 0.03 0.97 - - - - cmp si, -16192 - - - 0.05 0.94 - 0.01 - - mov ecx, 2 - - - 0.06 0.98 - 0.96 - - cmove rcx, rdx + - - 0.32 0.68 - - - - lea rcx, [rdi + rsi] - - - - - - - - xor eax, eax - - - 0.97 0.03 - - - - cmp esi, 49344 - - - 0.06 0.96 - 0.98 - - cmove rax, r8 - - - 0.02 0.03 - 0.95 - - mov rdx, rcx + - - 0.36 0.02 - 0.62 0.50 0.50 cmp word ptr [rdi + rsi], -16192 + - - 0.99 0.99 - 0.02 - - cmove rax, rcx - - - - - 1.00 - - ret diff --git a/zerocopy/src/lib.rs b/zerocopy/src/lib.rs index 3ac6db76fd..944f177c0b 100644 --- a/zerocopy/src/lib.rs +++ b/zerocopy/src/lib.rs @@ -1723,10 +1723,10 @@ runtime checks using `#[zerocopy(invariant(expression))]`: #[derive(TryFromBytes)] struct Foo { a: u8, - #[zerocopy(invariant((**a.unaligned_as_ref() % 2) == (**b.unaligned_as_ref() as u8)))] + #[zerocopy(invariant((*a.read() % 2) == (*b.read() as u8)))] b: bool, - #[zerocopy(invariant(**c.unaligned_as_ref() > 0))] - c: i8, + #[zerocopy(invariant(*c.read() > 0))] + c: i16, } ``` @@ -1734,7 +1734,9 @@ Each expression must return a `bool`. It has access to validated, read-only [`Ptr`]s to the current field and all preceding fields of the struct or variant, using their field names. A union's invariants have access only to the current field. The expression can use the existing [`Ptr`] APIs to -inspect those fields. In this example, all fields are accessed by reference. +inspect those fields. In this example, `read()` copies each field without +requiring alignment, and dereferencing the resulting [`ReadOnly`] accesses +the copied value. Rust's usual restrictions on local bindings apply; for example, a field name cannot shadow an in-scope constant. @@ -3281,17 +3283,7 @@ pub unsafe trait TryFromBytes { where Self: Sized, { - // FIXME(#2981): If `align_of::() == 1`, validate `source` in-place. - - let candidate = match CoreMaybeUninit::::read_from_bytes(source) { - Ok(candidate) => candidate, - Err(e) => { - return Err(TryReadError::Size(e.with_dst())); - } - }; - // SAFETY: `candidate` was copied from from `source: &[u8]`, so all of - // its bytes are initialized. - unsafe { try_read_from(source, candidate) } + try_read_from(source) } /// Attempts to read a `Self` from the prefix of the given `source`. @@ -3359,17 +3351,17 @@ pub unsafe trait TryFromBytes { where Self: Sized, { - // FIXME(#2981): If `align_of::() == 1`, validate `source` in-place. - - let (candidate, suffix) = match CoreMaybeUninit::::read_from_prefix(source) { - Ok(candidate) => candidate, - Err(e) => { - return Err(TryReadError::Size(e.with_dst())); - } + let (prefix, suffix) = match SplitAt::split_at(source, mem::size_of::()) { + Some(split) => split.via_immutable(), + None => return Err(SizeError::new(source).into()), }; - // SAFETY: `candidate` was copied from from `source: &[u8]`, so all of - // its bytes are initialized. - unsafe { try_read_from(source, candidate).map(|slf| (slf, suffix)) } + match try_read_from(prefix) { + Ok(slf) => Ok((slf, suffix)), + Err(e) => Err(e.map_src( + #[inline(always)] + |_| source, + )), + } } /// Attempts to read a `Self` from the suffix of the given `source`. @@ -3438,37 +3430,65 @@ pub unsafe trait TryFromBytes { where Self: Sized, { - // FIXME(#2981): If `align_of::() == 1`, validate `source` in-place. - - let (prefix, candidate) = match CoreMaybeUninit::::read_from_suffix(source) { - Ok(candidate) => candidate, - Err(e) => { - return Err(TryReadError::Size(e.with_dst())); - } + let split_at = match source.len().checked_sub(mem::size_of::()) { + Some(split_at) => split_at, + None => return Err(SizeError::new(source).into()), }; - // SAFETY: `candidate` was copied from from `source: &[u8]`, so all of - // its bytes are initialized. - unsafe { try_read_from(source, candidate).map(|slf| (prefix, slf)) } + // SAFETY: `checked_sub` returned the difference without overflow [1], + // so `split_at = source.len() - size_of::() <= source.len()`. + // This satisfies `SplitAt::split_at_unchecked`'s precondition. + // + // [1] Per https://doc.rust-lang.org/1.56.0/std/primitive.usize.html#method.checked_sub: + // + // Checked integer subtraction. Computes `self - rhs`, returning + // `None` if overflow occurred. + let (prefix, suffix) = + unsafe { SplitAt::split_at_unchecked(source, split_at) }.via_immutable(); + match try_read_from(suffix) { + Ok(slf) => Ok((prefix, slf)), + Err(e) => Err(e.map_src( + #[inline(always)] + |_| source, + )), + } } } +/// Validates and interprets the given affix of `source`'s bytes as a `&T`. +/// +/// Returns the destination reference and excess bytes on success. All errors, +/// including validity errors, contain the original `&S`. #[inline(always)] -fn try_ref_from_prefix_suffix( - source: &[u8], +fn try_ref_from_prefix_suffix( + source: &S, cast_type: CastType, meta: Option, -) -> Result<(&T, &[u8]), TryCastError<&[u8], T>> { - match Ptr::from_ref(source).try_cast_into::(cast_type, meta) { - Ok((source, prefix_suffix)) => { +) -> Result<(&T, &[u8]), TryCastError<&S, T>> +where + S: IntoBytes + Immutable + ?Sized, + T: TryFromBytes + KnownLayout + Immutable + ?Sized, +{ + match Ptr::from_ref(source.as_bytes()).try_cast_into::(cast_type, meta) { + Ok((candidate, prefix_suffix)) => { // This call may panic. If that happens, it doesn't cause any soundness // issues, as we have not generated any invalid state which we need to // fix before returning. - match source.try_into_safe() { + match candidate.try_into_safe() { Ok(valid) => Ok((valid.as_ref(), prefix_suffix.as_ref())), - Err(e) => Err(e.map_src(|src| src.as_bytes::().as_ref()).into()), + Err(e) => Err(e + .map_src( + #[inline(always)] + |_| source, + ) + .into()), } } - Err(e) => Err(e.map_src(Ptr::as_ref).into()), + Err(e) => Err(e + .map_src( + #[inline(always)] + |_| source, + ) + .into()), } } @@ -3497,49 +3517,112 @@ fn swap((t, u): (T, U)) -> (U, T) { (u, t) } -/// # Safety -/// -/// All bytes of `candidate` must be initialized. #[inline(always)] -unsafe fn try_read_from( - source: S, - mut candidate: CoreMaybeUninit, -) -> Result> { +fn try_read_from(source: &S) -> Result> +where + S: IntoBytes + Immutable + ?Sized, + T: TryFromBytes, +{ + let bytes = source.as_bytes(); + + if bytes.len() != mem::size_of::() { + return Err(SizeError::new(source).into()); + } + + // FIXME(#2981): Avoid the validation copy when validation can safely use + // a pointer derived from `source`'s shared borrow. + + // Initialize the candidate in its final location. A typed move of a + // `MaybeUninit` may discard initialized bytes at `T`'s padding offsets + // [1], but validation requires every byte to be initialized. Do not move + // `candidate` between this copy and validation. + // + // [1] Per https://doc.rust-lang.org/1.93.1/std/mem/union.MaybeUninit.html#validity: + // + // Moving or copying a value of type `MaybeUninit` (i.e., performing a + // "typed copy") will exactly preserve the contents, including the + // provenance, of all non-padding bytes of type `T` in the value's + // representation. + let mut candidate = CoreMaybeUninit::::uninit(); + + // SAFETY: The earlier `if bytes.len() != mem::size_of::()` returns on a + // size mismatch, so reaching this copy implies that `bytes.len()` equals + // `mem::size_of::()`. Copying that many `u8`s cannot overrun `bytes`. + // `candidate.as_mut_ptr()` is writable for the same number of bytes because + // `MaybeUninit` has `T`'s size. Both pointers are non-null and aligned + // for `u8`, including when `T` is zero-sized. The fresh local allocation + // cannot overlap `source`, which is an argument. Thus the read, write, + // alignment, and non-overlap requirements of [2] hold. Copying `u8`s + // initializes every destination byte, including those at `T`'s padding + // offsets. + // + // These writes cannot violate `candidate`'s bit validity because every bit + // pattern is valid for `MaybeUninit` [3], even if it is invalid for `T`. + // + // [2] Per https://doc.rust-lang.org/1.56.0/std/ptr/fn.copy_nonoverlapping.html: + // + // Copies `count * size_of::()` bytes from `src` to `dst`. The source + // and destination must *not* overlap. + // + // [3] Per https://doc.rust-lang.org/1.56.0/std/mem/union.MaybeUninit.html#layout: + // + // ... any bit value is valid for a `MaybeUninit` ... + unsafe { + ptr::copy_nonoverlapping( + bytes.as_ptr(), + candidate.as_mut_ptr().cast::(), + mem::size_of::(), + ); + } + // We use `from_mut` despite not mutating via `c_ptr` so that we don't need // to add a `T: Immutable` bound. let c_ptr = Ptr::from_mut(&mut candidate); + // SAFETY: `c_ptr` has no uninitialized sub-ranges because it derived from - // `candidate`, which the caller promises is entirely initialized. Since - // `candidate` is a `MaybeUninit`, it has no validity requirements, and so - // no values written to an `Initialized` `c_ptr` can violate its validity. - // Since `c_ptr` has `Exclusive` aliasing, no mutations may happen except - // via `c_ptr` so long as it is live, so we don't need to worry about the - // fact that `c_ptr` may have more restricted validity than `candidate`. + // `candidate`, whose bytes were all initialized by the copy above and which + // has not been moved since. Since `candidate` is a `MaybeUninit`, it has no + // validity requirements, and so no values written to an `Initialized` + // `c_ptr` can violate its validity. Since `c_ptr` has `Exclusive` aliasing, + // no mutations may happen except via `c_ptr` so long as it is live, so we + // don't need to worry about the fact that `c_ptr` may have more restricted + // validity than `candidate`. let c_ptr = unsafe { c_ptr.assume_validity::() }; - let mut c_ptr = c_ptr.cast::<_, crate::pointer::cast::CastSized, _>(); - // Since we don't have `T: KnownLayout`, we hack around that by using - // `Wrapping`, which implements `KnownLayout` even if `T` doesn't. + let c_ptr = c_ptr.cast::<_, crate::pointer::cast::CastSized, _>(); + + // SAFETY: `c_ptr` originated from a reference to `candidate`, so its + // address is aligned for `MaybeUninit`, which has `T`'s alignment [1]. + // `CastSized` preserves that address. `ReadOnly` is `repr(transparent)` + // with a single `T` field, so it also has `T`'s alignment [2], including + // when `T` is zero-sized. + // + // [1] Per https://doc.rust-lang.org/1.56.0/std/mem/union.MaybeUninit.html#layout: // + // `MaybeUninit` is guaranteed to have the same size, alignment, and + // ABI as `T`: + // + // [2] Per https://doc.rust-lang.org/1.93.1/reference/type-layout.html#the-transparent-representation: + // + // ... same layout and ABI as the only non-size 0 non-alignment 1 field, + // if present, or unit otherwise. + let mut c_ptr = unsafe { c_ptr.assume_alignment::() }; + // This call may panic. If that happens, it doesn't cause any soundness // issues, as we have not generated any invalid state which we need to fix // before returning. - if !Wrapping::::is_safe(c_ptr.reborrow_shared().forget_aligned()) { + if !T::is_safe(c_ptr.reborrow_shared()) { return Err(ValidityError::new(source).into()); } - fn _assert_same_size_and_validity() - where - Wrapping: pointer::TransmuteFrom, - T: pointer::TransmuteFrom, invariant::Safe, invariant::Safe>, - { - } - - _assert_same_size_and_validity::(); - - // SAFETY: We just validated that `candidate` contains a valid - // `Wrapping`, which has the same size and bit validity as `T`, as - // guaranteed by the preceding type assertion. + // SAFETY: `T::is_safe` returned true for `candidate`'s initialized bytes, + // so it contains a valid `T`, as required by [1]. Validation used a shared + // `ReadOnly` pointer, and the bytes have not been modified since. + // + // [1] Per https://doc.rust-lang.org/1.56.0/std/mem/union.MaybeUninit.html#method.assume_init: + // + // It is up to the caller to guarantee that the `MaybeUninit` really is + // in an initialized state. Ok(unsafe { candidate.assume_init() }) } @@ -5622,22 +5705,32 @@ pub unsafe trait FromBytes: FromZeros { } } -/// Interprets the given affix of the given bytes as a `&Self`. +/// Interprets the given affix of `source`'s bytes as a `&T`. /// -/// This method computes the largest possible size of `Self` that can fit in the -/// prefix or suffix bytes of `source`, then attempts to return both a reference -/// to those bytes interpreted as a `Self`, and a reference to the excess bytes. -/// If there are insufficient bytes, or if that affix of `source` is not -/// appropriately aligned, this returns `Err`. +/// This method uses `meta` if provided; otherwise, it computes the largest +/// possible size of `T` that can fit in the prefix or suffix bytes of `source`. +/// It returns both a reference to those bytes interpreted as a `T`, and a +/// reference to the excess bytes. If there are insufficient bytes, or if that +/// affix of `source` is not appropriately aligned, this returns `Err` containing +/// the original `&S`. #[inline(always)] -fn ref_from_prefix_suffix( - source: &[u8], +fn ref_from_prefix_suffix( + source: &S, meta: Option, cast_type: CastType, -) -> Result<(&T, &[u8]), CastError<&[u8], T>> { - let (slf, prefix_suffix) = Ptr::from_ref(source) +) -> Result<(&T, &[u8]), CastError<&S, T>> +where + S: IntoBytes + Immutable + ?Sized, + T: FromBytes + KnownLayout + Immutable + ?Sized, +{ + let (slf, prefix_suffix) = Ptr::from_ref(source.as_bytes()) .try_cast_into::<_, BecauseImmutable>(cast_type, meta) - .map_err(|err| err.map_src(|s| s.as_ref()))?; + .map_err(|err| { + err.map_src( + #[inline(always)] + |_| source, + ) + })?; Ok((slf.recall_validity().as_ref(), prefix_suffix.as_ref())) } @@ -7366,6 +7459,112 @@ mod tests { ); } + #[test] + fn test_try_read_from_typed_source() { + #[derive(IntoBytes, Immutable)] + #[repr(transparent)] + struct Source(T); + + fn check(source: &S, valid: bool) { + match try_read_from::<_, [bool; 4]>(source) { + Ok(value) => { + assert!(valid); + assert_eq!(value, [false, true, false, true]); + } + Err(error) => { + assert!(!valid); + assert!(matches!(error, TryReadError::Validity(_))); + assert!(ptr::eq(error.into_src(), source)); + } + } + + let error = try_read_from::<_, [bool; 5]>(source).unwrap_err(); + assert!(matches!(error, TryReadError::Size(_))); + assert!(ptr::eq(error.into_src(), source)); + } + + for (value, valid) in [(u16::from_ne_bytes([0, 1]), true), (0x0202, false)] { + let source = Source([value; 2]); + check(&source, valid); + let source: &Source<[u16]> = &source; + check(source, valid); + } + + let source = Source([(); 2]); + assert!(try_read_from::<_, ()>(&source).is_ok()); + let source: &Source<[()]> = &source; + assert!(try_read_from::<_, ()>(source).is_ok()); + } + + #[test] + fn test_try_read_initialized_padding() { + #[derive(KnownLayout, Immutable, Debug, PartialEq, Eq)] + #[repr(C, align(8))] + struct Padded(u8); + + // SAFETY: `Padded` has only a `u8` field and padding, so all + // initialized byte sequences are valid. Per + // https://doc.rust-lang.org/1.93.1/reference/behavior-considered-undefined.html#invalid-values: + // + // An integer (`i*`/`u*`), floating point value (`f*`), or raw pointer + // must be initialized, i.e., must not be obtained from uninitialized + // memory. + // + // A `struct`, tuple, and array requires all fields/elements to be + // valid at their respective type. + const _: () = unsafe { + unsafe_impl!(=> TryFromBytes for Padded; |candidate| { + // Observe every byte while validation is running, including + // padding that a typed move of `MaybeUninit` could + // discard. The returned `Padded` need not retain its padding. + assert_eq!(candidate.as_bytes::().as_ref(), &[0xA5; 8]); + true + }) + }; + + let source = [0xA5; 8]; + assert_eq!(Padded::try_read_from_bytes(&source), Ok(Padded(0xA5))); + + let prefix_source = [0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0]; + let (value, suffix) = Padded::try_read_from_prefix(&prefix_source).unwrap(); + assert_eq!(value, Padded(0xA5)); + assert!(ptr::eq(suffix, &prefix_source[8..])); + + let suffix_source = [0, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5, 0xA5]; + let (prefix, value) = Padded::try_read_from_suffix(&suffix_source).unwrap(); + assert_eq!(value, Padded(0xA5)); + assert!(ptr::eq(prefix, &suffix_source[..1])); + } + + #[test] + fn test_try_read_error_sources_and_zero_sized() { + let invalid = [2, 3]; + for (error, source) in [ + (bool::try_read_from_bytes(&invalid[..1]).unwrap_err(), &invalid[..1]), + (bool::try_read_from_prefix(&invalid).unwrap_err(), &invalid[..]), + (bool::try_read_from_suffix(&invalid).unwrap_err(), &invalid[..]), + ] { + assert!(matches!(error, TryReadError::Validity(_))); + assert!(ptr::eq(error.into_src(), source)); + } + for source in [&[][..], &invalid[..]] { + let error = bool::try_read_from_bytes(source).unwrap_err(); + assert!(matches!(error, TryReadError::Size(_))); + assert!(ptr::eq(error.into_src(), source)); + } + for error in [ + bool::try_read_from_prefix(&invalid[..0]).unwrap_err(), + bool::try_read_from_suffix(&invalid[..0]).unwrap_err(), + ] { + assert!(matches!(error, TryReadError::Size(_))); + assert!(ptr::eq(error.into_src(), &invalid[..0])); + } + + assert_eq!(<()>::try_read_from_bytes(&[]), Ok(())); + assert_eq!(<()>::try_read_from_prefix(&invalid), Ok(((), &invalid[..]))); + assert_eq!(<()>::try_read_from_suffix(&invalid), Ok((&invalid[..], ()))); + } + #[test] fn test_ref_from_mut_from_bytes() { // Test `FromBytes::{ref_from_bytes, mut_from_bytes}{,_prefix,Suffix}` @@ -7785,6 +7984,133 @@ mod tests { assert_eq!(rest.len(), 4); } + #[test] + fn test_ref_from_prefix_suffix_typed_source() { + #[derive(IntoBytes, Immutable)] + #[repr(transparent)] + struct Source(T); + + fn check(source: &S) { + let bytes = source.as_bytes(); + assert_eq!(bytes.len(), 4); + + for cast_type in [CastType::Prefix, CastType::Suffix] { + let (expected, expected_rest) = match cast_type { + CastType::Prefix => (&bytes[..3], &bytes[3..]), + CastType::Suffix => (&bytes[1..], &bytes[..1]), + }; + for meta in [None, Some(1)] { + let (value, rest) = + ref_from_prefix_suffix::<_, [[u8; 3]]>(source, meta, cast_type).unwrap(); + assert_eq!(value, &[expected]); + assert!(ptr::eq(value.as_bytes(), expected)); + assert!(ptr::eq(rest, expected_rest)); + } + + let err = + ref_from_prefix_suffix::<_, [u8; 5]>(source, None, cast_type).unwrap_err(); + assert!(matches!(err, CastError::Size(_))); + assert!(ptr::eq(err.into_src(), source)); + + let err = + ref_from_prefix_suffix::<_, [u8]>(source, Some(5), cast_type).unwrap_err(); + assert!(matches!(err, CastError::Size(_))); + assert!(ptr::eq(err.into_src(), source)); + } + } + + check(&Source([0x0102u16, 0x0304])); + let source = Source([false, true, false, true]); + let source: &Source<[bool]> = &source; + check(source); + } + + #[test] + fn test_ref_from_prefix_suffix_typed_source_alignment_error() { + let storage = Align::<[bool; 9], AU64>::new([false; 9]); + for cast_type in [CastType::Prefix, CastType::Suffix] { + let source = match cast_type { + CastType::Prefix => &storage.t[1..], + CastType::Suffix => &storage.t[..], + }; + let err = ref_from_prefix_suffix::<_, AU64>(source, None, cast_type).unwrap_err(); + assert!(matches!(err, CastError::Alignment(_))); + let recovered: &[bool] = err.into_src(); + assert!(ptr::eq(recovered, source)); + + let err = try_ref_from_prefix_suffix::<_, AU64>(source, cast_type, None).unwrap_err(); + assert!(matches!(err, TryCastError::Alignment(_))); + let recovered: &[bool] = err.into_src(); + assert!(ptr::eq(recovered, source)); + } + } + + #[test] + fn test_try_ref_from_prefix_suffix_typed_source() { + #[derive(IntoBytes, Immutable)] + #[repr(transparent)] + struct Source(T); + + fn check(source: &S) { + let bytes = source.as_bytes(); + assert_eq!(bytes, [0, 1, 0, 1]); + + for cast_type in [CastType::Prefix, CastType::Suffix] { + let (expected, expected_rest) = match cast_type { + CastType::Prefix => (&bytes[..3], &bytes[3..]), + CastType::Suffix => (&bytes[1..], &bytes[..1]), + }; + for meta in [None, Some(1)] { + let (value, rest) = + try_ref_from_prefix_suffix::<_, [[bool; 3]]>(source, cast_type, meta) + .unwrap(); + assert_eq!(value.as_bytes(), expected); + assert!(ptr::eq(value.as_bytes(), expected)); + assert!(ptr::eq(rest, expected_rest)); + } + + let err = try_ref_from_prefix_suffix::<_, [bool; 5]>(source, cast_type, None) + .unwrap_err(); + assert!(matches!(err, TryCastError::Size(_))); + assert!(ptr::eq(err.into_src(), source)); + + let err = try_ref_from_prefix_suffix::<_, [bool]>(source, cast_type, Some(5)) + .unwrap_err(); + assert!(matches!(err, TryCastError::Size(_))); + assert!(ptr::eq(err.into_src(), source)); + } + } + + check(&Source([u16::from_ne_bytes([0, 1]); 2])); + let source = Source([false, true, false, true]); + let source: &Source<[bool]> = &source; + check(source); + } + + #[test] + fn test_try_ref_from_prefix_suffix_typed_source_validity_error() { + fn check(source: &S) { + for cast_type in [CastType::Prefix, CastType::Suffix] { + let err = + try_ref_from_prefix_suffix::<_, bool>(source, cast_type, None).unwrap_err(); + assert!(matches!(err, TryCastError::Validity(_))); + assert!(ptr::eq(err.into_src(), source)); + + for meta in [None, Some(1)] { + let err = try_ref_from_prefix_suffix::<_, [bool]>(source, cast_type, meta) + .unwrap_err(); + assert!(matches!(err, TryCastError::Validity(_))); + assert!(ptr::eq(err.into_src(), source)); + } + } + } + + let source = [u16::from_ne_bytes([2, 2]); 2]; + check(&source); + check(&source[..]); + check(source.as_bytes()); + } + #[test] fn test_try_ref_from_prefix_suffix() { use crate::util::testutil::Align; diff --git a/zerocopy/src/wrappers.rs b/zerocopy/src/wrappers.rs index 4425ec26cb..2d4ff55159 100644 --- a/zerocopy/src/wrappers.rs +++ b/zerocopy/src/wrappers.rs @@ -8,7 +8,7 @@ // This file may not be copied, modified, or distributed except according to // those terms. -use core::{fmt, hash::Hash}; +use core::{borrow::Borrow, fmt, hash::Hash}; use super::*; use crate::pointer::{invariant::Safe, SizeEq, TransmuteFrom}; @@ -613,7 +613,14 @@ mod read_only_def { /// Note that `&mut ReadOnly` still permits mutation – the read-only /// property only applies to shared references. /// + /// `ReadOnly` implements [`Copy`] and [`Clone`] when `T: Copy`. Cloning + /// copies the value without calling `T::clone`. Trait implementations that + /// expose or operate on a shared reference to `T` require `T: Immutable`. + /// Constructing a wrapper or accessing its contents through [`AsMut`] + /// does not require `T: Immutable`. + /// /// [`Immutable`]: crate::Immutable + #[derive(Copy)] #[repr(transparent)] pub struct ReadOnly { // INVARIANT: `inner` is never mutated through a `&ReadOnly` @@ -715,6 +722,29 @@ unsafe impl TransmuteFrom for ReadOnly {} // it has the same bit validity as `T`. unsafe impl TransmuteFrom, Safe, Safe> for T {} +impl Default for ReadOnly { + #[inline(always)] + fn default() -> Self { + Self::new(Default::default()) + } +} + +// Copying avoids calling `T::clone`, which could mutate through a shared +// reference to `T` and violate `ReadOnly`'s invariant. +impl Clone for ReadOnly { + #[inline(always)] + fn clone(&self) -> Self { + *self + } +} + +impl From for ReadOnly { + #[inline(always)] + fn from(t: T) -> Self { + Self::new(t) + } +} + impl<'a, T: ?Sized + Immutable> From<&'a T> for &'a ReadOnly { #[inline(always)] fn from(t: &'a T) -> &'a ReadOnly { @@ -743,13 +773,72 @@ impl DerefMut for ReadOnly { } } -impl Debug for ReadOnly { +impl AsRef for ReadOnly { #[inline(always)] - fn fmt(&self, f: &mut Formatter<'_>) -> fmt::Result { - self.deref().fmt(f) + fn as_ref(&self) -> &T { + self.deref() + } +} + +impl Borrow for ReadOnly { + #[inline(always)] + fn borrow(&self) -> &T { + self.deref() + } +} + +impl AsMut for ReadOnly { + #[inline(always)] + fn as_mut(&mut self) -> &mut T { + ReadOnly::as_mut(self) + } +} + +// Delegate shared access through `Deref`, which requires `T: Immutable` and +// thus prevents interior mutation of the wrapped value by these trait methods. +impl PartialEq for ReadOnly { + #[inline(always)] + fn eq(&self, other: &Self) -> bool { + self.deref().eq(other.deref()) + } +} + +impl Eq for ReadOnly {} + +impl PartialOrd for ReadOnly { + #[inline(always)] + fn partial_cmp(&self, other: &Self) -> Option { + self.deref().partial_cmp(other.deref()) } } +impl Ord for ReadOnly { + #[inline(always)] + fn cmp(&self, other: &Self) -> Ordering { + self.deref().cmp(other.deref()) + } +} + +impl Hash for ReadOnly { + #[inline(always)] + fn hash(&self, state: &mut H) { + self.deref().hash(state); + } +} + +macro_rules! impl_read_only_fmt { + ($($trait:ident),* $(,)?) => {$( + impl fmt::$trait for ReadOnly { + #[inline(always)] + fn fmt(&self, f: &mut Formatter<'_>) -> fmt::Result { + fmt::$trait::fmt(self.deref(), f) + } + } + )*}; +} + +impl_read_only_fmt!(Debug, Display, Binary, Octal, LowerHex, UpperHex, LowerExp, UpperExp); + // SAFETY: See safety comment on `ProjectToTag`. unsafe impl + ?Sized, Client> HasTag for ReadOnly { #[allow(clippy::missing_inline_in_public_items)] @@ -857,6 +946,96 @@ mod tests { use super::*; use crate::util::testutil::*; + #[test] + #[allow(clippy::clone_on_copy, clippy::non_canonical_clone_impl)] + fn test_read_only_copy_clone() { + // This type deliberately does not implement `Immutable`. Copying the + // wrapper must not require it or invoke the inner `Clone` impl. + #[derive(Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] + struct PanickingClone(u16); + + impl Clone for PanickingClone { + fn clone(&self) -> Self { + panic!("ReadOnly must not call the inner Clone implementation") + } + } + + static_assertions::assert_not_impl_any!( + ReadOnly: Debug, PartialEq, Eq, PartialOrd, Ord, + Hash, AsRef, Borrow + ); + static_assertions::assert_not_impl_any!( + ReadOnly>: Copy, Clone, Deref, AsRef>, Borrow> + ); + + let value = ReadOnly::new(PanickingClone(42)); + let copy = value; + let clone = value.clone(); + assert_eq!(ReadOnly::into_inner(copy).0, 42); + assert_eq!(ReadOnly::into_inner(clone).0, 42); + } + + #[test] + fn test_read_only_mutable_access() { + // Construction and exclusive access also support interior mutability. + let mut value = ReadOnly::>::default(); + let inner: &mut Cell = value.as_mut(); + inner.set(42); + assert_eq!(ReadOnly::into_inner(value).get(), 42); + + let value = ReadOnly::from(Cell::new(7)); + assert_eq!(ReadOnly::into_inner(value).get(), 7); + } + + #[test] + fn test_read_only_borrowed_lookup() { + // Borrowed lookups require equality, hashing, and ordering to agree + // between the wrapper and the inner value. + let mut hash_map = std::collections::HashMap::new(); + hash_map.insert(ReadOnly::new(42u16), true); + assert_eq!(hash_map.get(&42u16), Some(&true)); + assert_eq!(hash_map.get(&43u16), None); + + let mut tree = std::collections::BTreeMap::new(); + tree.insert(ReadOnly::new(42u16), true); + assert_eq!(tree.get(&42u16), Some(&true)); + assert_eq!(tree.get(&43u16), None); + + // Preserve partial orders, including incomparable values. + let nan = ReadOnly::new(f32::NAN); + assert_eq!(nan.partial_cmp(&nan), None); + assert_ne!(nan, nan); + } + + #[test] + fn test_read_only_unsized() { + let bytes = [1u8, 2]; + let value: &ReadOnly<[u8]> = (&bytes[..]).into(); + let as_ref: &[u8] = value.as_ref(); + let borrowed: &[u8] = value.borrow(); + assert_eq!(as_ref, bytes); + assert_eq!(borrowed, bytes); + assert_eq!(value, value); + assert_eq!(value.cmp(value), Ordering::Equal); + + let text: &ReadOnly = "hello".into(); + assert_eq!(format!("{:.3}", text), "hel"); + } + + #[test] + fn test_read_only_formatting() { + let value = ReadOnly::new(42u16); + // Check that formatting flags are forwarded as well as the value. + assert_eq!(format!("{:04?}", value), "0042"); + assert_eq!(format!("{:04}", value), "0042"); + assert_eq!(format!("{:#010b}", value), "0b00101010"); + assert_eq!(format!("{:#06o}", value), "0o0052"); + assert_eq!(format!("{:#06x}", value), "0x002a"); + assert_eq!(format!("{:#06X}", value), "0x002A"); + assert_eq!(format!("{:.2e}", value), "4.20e1"); + assert_eq!(format!("{:.2E}", value), "4.20E1"); + } + #[test] fn test_unalign() { // Test methods that don't depend on alignment. diff --git a/zerocopy/zerocopy-derive/tests/invariant.rs b/zerocopy/zerocopy-derive/tests/invariant.rs index e4a63ca24d..12bf65ace3 100644 --- a/zerocopy/zerocopy-derive/tests/invariant.rs +++ b/zerocopy/zerocopy-derive/tests/invariant.rs @@ -30,9 +30,9 @@ fn read( #[repr(C, packed)] struct Foo { a: u8, - #[zerocopy(invariant((read(a) % 2) == (read(b) as u8)))] + #[zerocopy(invariant((*a.read() % 2) == (*b.read() as u8)))] b: bool, - #[zerocopy(invariant(read(c) > 0))] + #[zerocopy(invariant(*c.read() > 0))] c: i16, }