Raw runs: data/results/amd-cpu/cpu/session_matrix_aa/20260929_133603
Command: BENCH_AA_ROUNDS=4 BENCH_AA_WARMUPS=3 BENCH_AA_RUNS=15 BENCH_AA_THREADS='1 4' BENCH_COVERAGE=quick BENCHMARK_TARGET_PROFILE=amd-cpu TENFERRO_CPU_FEATURES=system-openblas scripts/run_paired_timing.sh aa /home/shinaoka/tensor4all/tenferro-rs/.worktrees/bench-main
Same tenferro-rs build in both arms, balanced order. Use these spreads to declare noise.max_aa_relative_spread, noise.max_cov and the relative threshold in the confirmation config before any candidate run. No verdicts are produced here.
| Case | A/A statistic | Spread | Max CoV | Rounds | Round ratios |
|---|---|---|---|---|---|
bdot_f64_b1024_m4n4k4_canonical_alloc_auto[faer]@t1 |
0.8942 | 0.1058 | 0.0112 | 4 | 0.7864, 0.7827, 1.0020, 1.2814 |
bdot_f64_b1024_m4n4k4_canonical_alloc_auto[faer]@t4 |
1.0003 | 0.0003 | 0.0116 | 4 | 1.2810, 0.9993, 1.0002, 1.0004 |
bdot_f64_b1024_m4n4k4_canonical_into_auto[faer]@t1 |
0.9958 | 0.0042 | 0.0089 | 4 | 1.0035, 1.0017, 0.9900, 0.9866 |
bdot_f64_b1024_m4n4k4_canonical_into_auto[faer]@t4 |
0.9874 | 0.0126 | 0.1386 | 4 | 0.9677, 1.0128, 0.8391, 1.0071 |
bdot_f64_b1024_m4n4k4_direct_alloc_auto[faer]@t1 |
1.0014 | 0.0014 | 0.0929 | 4 | 0.9714, 0.9992, 1.0036, 1.1872 |
bdot_f64_b1024_m4n4k4_direct_alloc_auto[faer]@t4 |
0.9975 | 0.0025 | 0.2404 | 4 | 1.1042, 0.9938, 1.0008, 0.9942 |
bdot_f64_b1024_m4n4k4_direct_into_auto[faer]@t1 |
1.0030 | 0.0030 | 0.0727 | 4 | 0.9938, 0.9924, 1.0122, 1.0174 |
bdot_f64_b1024_m4n4k4_direct_into_auto[faer]@t4 |
0.9400 | 0.0600 | 0.0900 | 4 | 0.9213, 0.9993, 0.9587, 0.8994 |
bdot_f64_b16_m64n64k64_direct_alloc_auto[faer]@t1 |
1.0001 | 0.0001 | 0.0916 | 4 | 0.8366, 0.9981, 1.0056, 1.0021 |
bdot_f64_b16_m64n64k64_direct_alloc_auto[faer]@t4 |
1.0005 | 0.0005 | 0.0109 | 4 | 0.9981, 1.0092, 0.9934, 1.0029 |
bdot_f64_b16_m64n64k64_direct_into_auto[faer]@t1 |
1.0063 | 0.0063 | 0.0663 | 4 | 0.9957, 1.0114, 1.2144, 1.0012 |
bdot_f64_b16_m64n64k64_direct_into_auto[faer]@t4 |
0.8004 | 0.1996 | 0.2313 | 4 | 1.2965, 0.5840, 1.0168, 0.5037 |
bdot_f64_b64_m4n4k4_direct_alloc_auto[faer]@t1 |
1.0233 | 0.0233 | 0.0078 | 4 | 1.0000, 1.0655, 0.9567, 1.0465 |
bdot_f64_b64_m4n4k4_direct_alloc_auto[faer]@t4 |
1.0084 | 0.0084 | 0.0271 | 4 | 1.0114, 1.0092, 1.0076, 0.9977 |
bdot_f64_b64_m4n4k4_direct_into_auto[faer]@t1 |
0.9993 | 0.0007 | 0.0110 | 4 | 1.0275, 0.9859, 1.0010, 0.9975 |
bdot_f64_b64_m4n4k4_direct_into_auto[faer]@t4 |
1.0033 | 0.0033 | 0.0978 | 4 | 0.8027, 0.9974, 1.0092, 1.0177 |
bdot_f64_b64_m64n1k64_direct_alloc_auto[faer]@t1 |
1.0264 | 0.0264 | 0.0752 | 4 | 1.0217, 1.0311, 1.0448, 1.0141 |
bdot_f64_b64_m64n1k64_direct_alloc_auto[faer]@t4 |
0.8888 | 0.1112 | 0.1008 | 4 | 0.8161, 0.7775, 0.9615, 1.0478 |
bdot_f64_b64_m64n1k64_direct_into_auto[faer]@t1 |
1.0075 | 0.0075 | 0.0236 | 4 | 1.0348, 0.9599, 1.0363, 0.9801 |
bdot_f64_b64_m64n1k64_direct_into_auto[faer]@t4 |
1.0656 | 0.0656 | 0.1041 | 4 | 1.0842, 0.9640, 1.1700, 1.0470 |
beinsum_f64_b1024_m4n4k4_direct_alloc_auto[faer]@t1 |
0.9968 | 0.0032 | 0.0521 | 4 | 1.0056, 1.0006, 0.9929, 0.9820 |
beinsum_f64_b1024_m4n4k4_direct_alloc_auto[faer]@t4 |
1.0055 | 0.0055 | 0.0236 | 4 | 1.0110, 1.0000, 1.0138, 0.9945 |
beinsum_f64_b1024_m4n4k4_direct_into_auto[faer]@t1 |
1.0058 | 0.0058 | 0.0495 | 4 | 1.0107, 1.0094, 0.9968, 1.0022 |
beinsum_f64_b1024_m4n4k4_direct_into_auto[faer]@t4 |
1.0054 | 0.0054 | 0.1048 | 4 | 1.0018, 1.0767, 1.0090, 0.9862 |
chain3_f64_n4_einsum_alloc[faer]@t1 |
0.9959 | 0.0041 | 0.4051 | 4 | 1.0178, 0.9782, 1.0137, 0.9162 |
chain3_f64_n4_einsum_alloc[faer]@t4 |
1.0172 | 0.0172 | 0.0518 | 4 | 1.0300, 0.8407, 1.0043, 1.2198 |
hadamard_f64_m64n64_dot_alloc[faer]@t1 |
0.9412 | 0.0588 | 0.0900 | 4 | 0.9979, 0.9921, 0.8309, 0.8902 |
hadamard_f64_m64n64_dot_alloc[faer]@t4 |
1.0026 | 0.0026 | 0.0048 | 4 | 1.0010, 1.0072, 1.0042, 0.9964 |
hadamard_f64_m64n64_einsum_alloc[faer]@t1 |
0.9547 | 0.0453 | 0.0901 | 4 | 0.9807, 0.9288, 0.9157, 0.9925 |
hadamard_f64_m64n64_einsum_alloc[faer]@t4 |
0.9937 | 0.0063 | 0.0237 | 4 | 0.9973, 1.0150, 0.9902, 0.9826 |
matmul_n16_count1024[blas]@t1 |
1.0026 | 0.0026 | 0.0875 | 4 | 0.9981, 1.0007, 1.1173, 1.0045 |
matmul_n16_count1024[blas]@t4 |
0.9976 | 0.0024 | 0.0832 | 4 | 0.9948, 1.0005, 0.9780, 1.0043 |
matmul_n16_count1024[faer]@t1 |
0.9705 | 0.0295 | 0.0907 | 4 | 0.9467, 0.9335, 0.9942, 1.0698 |
matmul_n16_count1024[faer]@t4 |
0.9917 | 0.0083 | 0.0135 | 4 | 0.9917, 0.9862, 0.9916, 1.0031 |
matmul_n16_count1024[pytorch]@t1 |
1.0070 | 0.0070 | 0.0885 | 4 | 1.0051, 1.0117, 1.0090, 0.9892 |
matmul_n16_count1024[pytorch]@t4 |
0.9324 | 0.0676 | 0.1198 | 4 | 0.8212, 1.0152, 0.8781, 0.9867 |
matmul_n2_count1024[blas]@t1 |
1.0010 | 0.0010 | 0.0916 | 4 | 0.9975, 1.1977, 0.9333, 1.0046 |
matmul_n2_count1024[blas]@t4 |
0.9984 | 0.0016 | 0.0102 | 4 | 1.0028, 0.9975, 0.9756, 0.9994 |
matmul_n2_count1024[faer]@t1 |
0.9998 | 0.0002 | 0.0643 | 4 | 0.9913, 1.0512, 0.9984, 1.0012 |
matmul_n2_count1024[faer]@t4 |
0.9049 | 0.0951 | 0.2829 | 4 | 1.0033, 0.6888, 1.0140, 0.8066 |
matmul_n2_count1024[pytorch]@t1 |
0.9951 | 0.0049 | 0.1422 | 4 | 0.9969, 1.0341, 0.9934, 0.9884 |
matmul_n2_count1024[pytorch]@t4 |
0.9835 | 0.0165 | 0.0853 | 4 | 0.9862, 0.9458, 0.9808, 1.0276 |
matmul_n32_count1024[blas]@t1 |
0.9567 | 0.0433 | 0.0758 | 4 | 0.8926, 0.9449, 0.9992, 0.9685 |
matmul_n32_count1024[blas]@t4 |
0.9901 | 0.0099 | 0.0646 | 4 | 0.9974, 1.0040, 0.9633, 0.9827 |
matmul_n32_count1024[faer]@t1 |
1.0116 | 0.0116 | 0.0795 | 4 | 0.9032, 1.0921, 1.0277, 0.9955 |
matmul_n32_count1024[faer]@t4 |
0.9643 | 0.0357 | 0.0660 | 4 | 0.9625, 0.9661, 0.9534, 1.1301 |
matmul_n32_count1024[pytorch]@t1 |
1.0179 | 0.0179 | 0.1445 | 4 | 1.0206, 0.9031, 1.0229, 1.0152 |
matmul_n32_count1024[pytorch]@t4 |
0.9806 | 0.0194 | 0.0629 | 4 | 1.0087, 0.9502, 0.9976, 0.9635 |
matmul_n4_count1024[blas]@t1 |
0.9225 | 0.0775 | 0.0941 | 4 | 0.8302, 0.8547, 1.0029, 0.9904 |
matmul_n4_count1024[blas]@t4 |
0.9869 | 0.0131 | 0.0106 | 4 | 0.9748, 0.9908, 0.9963, 0.9831 |
matmul_n4_count1024[faer]@t1 |
1.0046 | 0.0046 | 0.0811 | 4 | 0.9980, 1.0458, 0.9991, 1.0101 |
matmul_n4_count1024[faer]@t4 |
0.9976 | 0.0024 | 0.0112 | 4 | 0.9813, 0.9974, 1.0071, 0.9978 |
matmul_n4_count1024[pytorch]@t1 |
0.9731 | 0.0269 | 0.1187 | 4 | 0.9694, 0.8734, 0.9767, 1.0093 |
matmul_n4_count1024[pytorch]@t4 |
1.0017 | 0.0017 | 0.0638 | 4 | 1.0171, 0.9733, 1.0298, 0.9863 |
matmul_n8_count1024[blas]@t1 |
0.8423 | 0.1577 | 0.0887 | 4 | 0.8376, 0.8440, 0.8407, 1.0071 |
matmul_n8_count1024[blas]@t4 |
0.9949 | 0.0051 | 0.0188 | 4 | 0.9960, 0.9996, 0.9356, 0.9939 |
matmul_n8_count1024[faer]@t1 |
0.9624 | 0.0376 | 0.0909 | 4 | 0.9791, 0.9457, 0.9437, 1.2166 |
matmul_n8_count1024[faer]@t4 |
0.9998 | 0.0002 | 0.0088 | 4 | 0.9978, 1.0023, 0.9668, 1.0017 |
matmul_n8_count1024[pytorch]@t1 |
0.9989 | 0.0011 | 0.0905 | 4 | 1.0054, 0.9887, 1.0390, 0.9925 |
matmul_n8_count1024[pytorch]@t4 |
1.0165 | 0.0165 | 0.1351 | 4 | 0.8175, 1.0792, 1.0640, 0.9690 |
solve_n16_count1024[blas]@t1 |
0.9976 | 0.0024 | 0.0783 | 4 | 0.8343, 0.9929, 1.0042, 1.0024 |
solve_n16_count1024[blas]@t4 |
1.0066 | 0.0066 | 0.0653 | 4 | 0.9543, 1.0049, 1.0587, 1.0082 |
solve_n16_count1024[faer]@t1 |
0.8950 | 0.1050 | 0.0807 | 4 | 0.8559, 0.9342, 0.8391, 1.0045 |
solve_n16_count1024[faer]@t4 |
1.0044 | 0.0044 | 0.0448 | 4 | 0.9925, 1.0043, 1.0046, 1.0554 |
solve_n16_count1024[pytorch]@t1 |
0.9641 | 0.0359 | 0.0887 | 4 | 0.8311, 0.9341, 0.9942, 1.0191 |
solve_n16_count1024[pytorch]@t4 |
1.0026 | 0.0026 | 0.0776 | 4 | 1.0067, 1.0545, 0.9682, 0.9984 |
solve_n2_count1024[blas]@t1 |
0.9893 | 0.0107 | 0.0926 | 4 | 1.2004, 0.9772, 1.0014, 0.8379 |
solve_n2_count1024[blas]@t4 |
1.0552 | 0.0552 | 0.0576 | 4 | 1.0883, 1.0238, 1.0867, 1.0089 |
solve_n2_count1024[faer]@t1 |
0.9265 | 0.0735 | 0.0861 | 4 | 1.0508, 0.9338, 0.8418, 0.9193 |
solve_n2_count1024[faer]@t4 |
0.9906 | 0.0094 | 0.1762 | 4 | 0.9925, 0.9887, 0.9748, 0.9957 |
solve_n2_count1024[pytorch]@t1 |
0.9792 | 0.0208 | 0.0805 | 4 | 0.9829, 0.9755, 0.9897, 0.9541 |
solve_n2_count1024[pytorch]@t4 |
0.9896 | 0.0104 | 0.0752 | 4 | 0.8304, 0.9895, 1.0006, 0.9897 |
solve_n32_count1024[blas]@t1 |
0.9875 | 0.0125 | 0.0789 | 4 | 0.8236, 0.9993, 0.9779, 0.9971 |
solve_n32_count1024[blas]@t4 |
0.9978 | 0.0022 | 0.0527 | 4 | 0.9686, 1.0172, 0.9879, 1.0078 |
solve_n32_count1024[faer]@t1 |
0.9941 | 0.0059 | 0.0706 | 4 | 0.9489, 1.0845, 0.9880, 1.0001 |
solve_n32_count1024[faer]@t4 |
0.9986 | 0.0014 | 0.0788 | 4 | 0.9983, 1.0329, 0.9647, 0.9989 |
solve_n32_count1024[pytorch]@t1 |
0.9999 | 0.0001 | 0.0624 | 4 | 0.8401, 1.0504, 1.0092, 0.9906 |
solve_n32_count1024[pytorch]@t4 |
0.9947 | 0.0053 | 0.0485 | 4 | 0.9909, 1.0243, 0.9863, 0.9986 |
solve_n4_count1024[blas]@t1 |
1.1207 | 0.1207 | 0.0864 | 4 | 1.0526, 0.8916, 1.1888, 1.2096 |
solve_n4_count1024[blas]@t4 |
1.0179 | 0.0179 | 0.0647 | 4 | 1.0127, 1.0231, 0.9052, 1.1266 |
solve_n4_count1024[faer]@t1 |
0.9984 | 0.0016 | 0.0846 | 4 | 0.9954, 0.8317, 1.0014, 1.1948 |
solve_n4_count1024[faer]@t4 |
0.9970 | 0.0030 | 0.2273 | 4 | 1.0023, 0.9674, 1.0065, 0.9917 |
solve_n4_count1024[pytorch]@t1 |
0.9986 | 0.0014 | 0.0765 | 4 | 0.9806, 1.0045, 0.9973, 0.9999 |
solve_n4_count1024[pytorch]@t4 |
1.0039 | 0.0039 | 0.0569 | 4 | 0.8544, 1.0071, 1.0006, 1.0108 |
solve_n8_count1024[blas]@t1 |
1.0083 | 0.0083 | 0.0817 | 4 | 1.0058, 1.0122, 0.8299, 1.0108 |
solve_n8_count1024[blas]@t4 |
1.0439 | 0.0439 | 0.0572 | 4 | 1.0752, 1.1147, 1.0127, 0.9965 |
solve_n8_count1024[faer]@t1 |
0.9348 | 0.0652 | 0.0922 | 4 | 0.8731, 0.8475, 0.9964, 1.2036 |
solve_n8_count1024[faer]@t4 |
0.9845 | 0.0155 | 0.0184 | 4 | 0.9845, 1.0100, 0.9846, 0.9834 |
solve_n8_count1024[pytorch]@t1 |
0.9656 | 0.0344 | 0.0650 | 4 | 0.8483, 0.9634, 0.9679, 1.0018 |
solve_n8_count1024[pytorch]@t4 |
0.9804 | 0.0196 | 0.0702 | 4 | 0.8675, 0.9808, 0.9801, 0.9978 |
stream_f64_fixed_len32_einsum_alloc[faer]@t1 |
1.0164 | 0.0164 | 0.0201 | 4 | 1.0428, 1.0039, 1.0289, 1.0032 |
stream_f64_fixed_len32_einsum_alloc[faer]@t4 |
0.9875 | 0.0125 | 0.0950 | 4 | 0.9957, 1.0062, 0.8028, 0.9792 |
stream_f64_fresh_len32_einsum_alloc[faer]@t1 |
0.9910 | 0.0090 | 0.0260 | 4 | 0.9171, 0.9967, 1.0205, 0.9852 |
stream_f64_fresh_len32_einsum_alloc[faer]@t4 |
0.9951 | 0.0049 | 0.0873 | 4 | 1.0019, 0.9852, 0.9883, 1.0137 |
stream_f64_mixed_len32_einsum_alloc[faer]@t1 |
0.9987 | 0.0013 | 0.0214 | 4 | 1.0720, 0.9861, 0.9985, 0.9989 |
stream_f64_mixed_len32_einsum_alloc[faer]@t4 |
0.9924 | 0.0076 | 0.0421 | 4 | 1.0363, 0.9881, 0.9966, 0.9802 |
stream_f64_strides_len32_einsum_alloc[faer]@t1 |
1.0064 | 0.0064 | 0.0292 | 4 | 1.0446, 0.9864, 0.9985, 1.0143 |
stream_f64_strides_len32_einsum_alloc[faer]@t4 |
1.0223 | 0.0223 | 0.0896 | 4 | 1.0448, 1.0263, 1.0173, 1.0182 |