refine benchmarks

This commit is contained in:
Marc Marcos 2026-06-12 09:00:52 +02:00
parent f639be16a0
commit 781601fe9e
15 changed files with 218279 additions and 123512 deletions

View file

@ -52,31 +52,43 @@ def parse_cycle_deltas(file_path):
if __name__ == "__main__":
parser = argparse.ArgumentParser(description='Extract cycle deltas from trace files and calculate speedup')
parser.add_argument('with_file', help='Path to the trace file (with optimization)')
parser.add_argument('without_file', help='Path to the trace file (without optimization - baseline)')
parser.add_argument('files', nargs='+', help='Path to trace file(s). Last file is baseline if more than one.')
args = parser.parse_args()
try:
df_with = parse_cycle_deltas(args.with_file)
df_without = parse_cycle_deltas(args.without_file)
if len(args.files) == 2:
files = args.files
baseline_idx = 1
elif len(args.files) >= 3:
files = args.files
baseline_idx = len(files) - 1
else:
print("Error: At least 2 files required")
exit(1)
if len(df_with) != len(df_without):
raise ValueError(
f"Mismatch in number of deltas: {args.with_file} has {len(df_with)} deltas, "
f"but {args.without_file} has {len(df_without)} deltas. Cannot pair them."
)
try:
dfs = [parse_cycle_deltas(f) for f in files]
baseline_df = dfs[baseline_idx]
for i, df in enumerate(dfs):
if len(df) != len(baseline_df):
raise ValueError(
f"Mismatch in number of deltas: {files[i]} has {len(df)} deltas, "
f"but {files[baseline_idx]} has {len(baseline_df)} deltas. Cannot pair them."
)
result = pd.DataFrame({
'start_cycle_with': df_with['start_cycle'],
'end_cycle_with': df_with['end_cycle'],
'delta_with': df_with['delta'],
'start_cycle_without': df_without['start_cycle'],
'end_cycle_without': df_without['end_cycle'],
'delta_without': df_without['delta'],
'speedup': df_without['delta'] / df_with['delta']
'start_cycle_baseline': baseline_df['start_cycle'],
'end_cycle_baseline': baseline_df['end_cycle'],
'delta_baseline': baseline_df['delta']
})
print("Cycle Delta Analysis:")
for i, df in enumerate(dfs):
if i == baseline_idx:
result[f'speedup_{i}'] = 1.0
else:
result[f'speedup_{i}'] = baseline_df['delta'] / df['delta']
print("Cycle Delta Analysis (baseline: " + files[baseline_idx] + "):")
print(result)
result.to_parquet("result.parquet")

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

20184
traces/with_zvfhmin.txt Normal file

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

24122
traces/without_zvfhmin.txt Normal file

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

View file

@ -6,6 +6,30 @@
#define ZFA
int32_t emulate_fcvtmod_w_d(double a) {
// 1. Trap NaNs and Infinities (Spec says they must return 0)
if (!isfinite(a)) {
return 0;
}
// 2. Bring the double into a range that fits in a 64-bit int.
// We use fmod to strip away everything above 2^32.
// (2^32 is exactly 4294967296.0)
double truncated_range = fmod(a, 4294967296.0);
// 3. Convert to 64-bit integer with truncation (RTZ)
int64_t standard_int = (int64_t)truncated_range;
// 4. Cast to 32-bit signed integer (effectively your SLLI/SRAI)
return (int32_t)standard_int;
}
static int baremetal_errno = 0;
int *__errno(void) {
return &baremetal_errno;
}
static inline uint64_t read_cycles() {
uint64_t start;
asm volatile ("rdcycle %0" : "=r"(start));
@ -13,197 +37,273 @@ static inline uint64_t read_cycles() {
}
int main() {
// fround.s
volatile float a;
volatile float b;
volatile double c;
volatile double d;
// fround.s
read_cycles();
for (int i = 0; i < N; i++) {
a = 3.25f;
b = round(a);
// -- fround.s --
{
volatile float a, b;
read_cycles();
for (int i = 0; i < N; i++) {
a = 3.25f;
b = round(a);
}
read_cycles();
}
read_cycles();
// fround.d
read_cycles();
for (int i = 0; i < N; i++) {
c = 3.25f;
d = round(c);
// -- fround.d --
{
volatile double a, b;
read_cycles();
for (int i = 0; i < N; i++) {
a = 3.25;
b = round(a);
}
read_cycles();
}
read_cycles();
int res;
// fleq.s
read_cycles();
for (int i = 0; i < N; i++) {
#ifndef ZFA
asm volatile (
"fclass.s t0, %1\n\t" // Classify a
"fclass.s t1, %2\n\t" // Classify b
"or t0, t0, t1\n\t" // Combine classes
"andi t2, t0, 0x200\n\t" // 0x200 is the mask for Quiet NaN
"bnez t2, 1f\n\t" // If qNaN is present, skip to return 0
"fle.s %0, %1, %2\n\t" // Safe to use signaling comparison
"j 2f\n\t"
"1:\n\t"
"li %0, 0\n\t" // Result is false for NaNs
"2:\n\t"
: "=r" (res)
: "f" (a), "f" (b)
: "t0", "t1", "t2"
);
#else
asm volatile("fleq.s t0, ft0, ft1");
#endif
// -- fleq.s --
{
volatile float a = 3.25f, b = 3.0f;
int res;
read_cycles();
for (int i = 0; i < N; i++) {
#ifndef ZFA
asm volatile (
"fclass.s t0, %1\n\t"
"fclass.s t1, %2\n\t"
"or t0, t0, t1\n\t"
"andi t2, t0, 0x200\n\t"
"bnez t2, 1f\n\t"
"fle.s %0, %1, %2\n\t"
"j 2f\n\t"
"1:\n\t"
"li %0, 0\n\t"
"2:\n\t"
: "=r" (res)
: "f" (a), "f" (b)
: "t0", "t1", "t2"
);
#else
asm volatile ("fleq.s %0, %1, %2" : "=r" (res) : "f" (a), "f" (b));
#endif
}
read_cycles();
}
read_cycles();
// fleq.d
read_cycles();
for (int i = 0; i < N; i++) {
#ifndef ZFA
asm volatile (
"fclass.d t0, %1\n\t" // Classify double a
"fclass.d t1, %2\n\t" // Classify double b
"or t0, t0, t1\n\t" // Combine classification masks
"andi t2, t0, 0x200\n\t" // 0x200 is the bit for Quiet NaN (qNaN)
"bnez t2, 1f\n\t" // If a qNaN is detected, skip to return 0
"fle.d %0, %1, %2\n\t" // Signaling comparison: signals on sNaN, result in %0
"j 2f\n\t"
"1:\n\t"
"li %0, 0\n\t" // Quietly return 0 (false) for qNaNs
"2:\n\t"
: "=r" (res)
: "f" (a), "f" (b)
: "t0", "t1", "t2"
);
#else
asm volatile ("fleq.d t0, ft0, ft1");
#endif
// -- fleq.d --
{
volatile double a = 3.25, b = 3.0;
int res;
read_cycles();
for (int i = 0; i < N; i++) {
#ifndef ZFA
asm volatile (
"fclass.d t0, %1\n\t"
"fclass.d t1, %2\n\t"
"or t0, t0, t1\n\t"
"andi t2, t0, 0x200\n\t"
"bnez t2, 1f\n\t"
"fle.d %0, %1, %2\n\t"
"j 2f\n\t"
"1:\n\t"
"li %0, 0\n\t"
"2:\n\t"
: "=r" (res)
: "f" (a), "f" (b)
: "t0", "t1", "t2"
);
#else
asm volatile ("fleq.d %0, %1, %2" : "=r" (res) : "f" (a), "f" (b));
#endif
}
read_cycles();
}
read_cycles();
// fminm.s
float a_fmin = 0.0f, b_fmin = -0.0f;
float res_fmin;
read_cycles();
for (int i = 0; i < N; i++) {
#ifndef ZFA
asm volatile (
"fclass.s t0, %1\n\t" // Classify a
"fclass.s t1, %2\n\t" // Classify b
"li t2, 0x300\n\t" // Mask for any NaN (0x100 sNaN | 0x200 qNaN)
"and t3, t0, t2\n\t" // t3 = is_nan(a)
"and t4, t1, t2\n\t" // t4 = is_nan(b)
"bnez t3, 1f\n\t" // If a is NaN, jump to handle it
"bnez t4, 2f\n\t" // If b is NaN, jump to handle it
"fmin.s %0, %1, %2\n\t" // Neither is NaN, use standard min
"j 3f\n\t"
"1:\n\t" // Case: a is NaN
"bnez t4, 4f\n\t" // If b is also NaN, jump to both-NaN case
"fmv.s %0, %2\n\t" // a is NaN, b is number -> return b
"j 3f\n\t"
"2:\n\t" // Case: b is NaN, a is number -> return a
"fmv.s %0, %1\n\t"
"j 3f\n\t"
"4:\n\t" // Case: Both are NaNs
"fmin.s %0, %1, %2\n\t" // Standard min handles both-NaNs correctly
"3:\n\t"
: "=f" (res_fmin)
: "f" (a_fmin), "f" (b_fmin)
: "t0", "t1", "t2", "t3", "t4"
);
#else
asm volatile ("fminm.s ft0, ft1, ft2");
#endif
// -- fminm.s --
{
volatile float a = 0.0f, b = -0.0f, res;
read_cycles();
for (int i = 0; i < N; i++) {
#ifndef ZFA
asm volatile (
"fclass.s t0, %1\n\t"
"fclass.s t1, %2\n\t"
"li t2, 0x300\n\t"
"and t3, t0, t2\n\t"
"and t4, t1, t2\n\t"
"bnez t3, 1f\n\t"
"bnez t4, 2f\n\t"
"fmin.s %0, %1, %2\n\t"
"j 3f\n\t"
"1:\n\t"
"bnez t4, 4f\n\t"
"fmv.s %0, %2\n\t"
"j 3f\n\t"
"2:\n\t"
"fmv.s %0, %1\n\t"
"j 3f\n\t"
"4:\n\t"
"fmin.s %0, %1, %2\n\t"
"3:\n\t"
: "=f" (res)
: "f" (a), "f" (b)
: "t0", "t1", "t2", "t3", "t4"
);
#else
asm volatile ("fminm.s %0, %1, %2" : "=f" (res) : "f" (a), "f" (b));
#endif
}
read_cycles();
}
read_cycles();
// fli.s
read_cycles();
volatile float res_fli_s[32];
for (int i = 0; i < N; i++) {
res_fli_s[0] = -1.0f;
res_fli_s[1] = -1.0f;
res_fli_s[2] = 0x1p-16f;
res_fli_s[3] = 0x1p-15f;
res_fli_s[4] = 0x1p-14f;
res_fli_s[5] = 0x1p-13f;
res_fli_s[6] = 0x1p-12f;
res_fli_s[7] = 0x1p-11f;
res_fli_s[8] = 0x1p-10f;
res_fli_s[9] = 0x1p-9f;
res_fli_s[10] = 0x1p-8f;
res_fli_s[11] = 0x1p-7f;
res_fli_s[12] = 0x1p-6f;
res_fli_s[13] = 0x1p-5f;
res_fli_s[14] = 0x1p-4f;
res_fli_s[15] = 0x1p-3f;
res_fli_s[16] = 0.25f;
res_fli_s[17] = 0.5f;
res_fli_s[18] = 0.75f;
res_fli_s[19] = 1.0f;
res_fli_s[20] = 1.25f;
res_fli_s[21] = 1.5f;
res_fli_s[22] = 1.75f;
res_fli_s[23] = 2.0f;
res_fli_s[24] = 2.5f;
res_fli_s[25] = 3.0f;
res_fli_s[26] = 4.0f;
res_fli_s[27] = 8.0f;
res_fli_s[28] = 16.0f;
res_fli_s[29] = 32.0f;
res_fli_s[30] = INFINITY;
res_fli_s[31] = NAN;
// -- fminm.d --
{
volatile double a = 0.0, b = -0.0, res;
read_cycles();
for (int i = 0; i < N; i++) {
#ifndef ZFA
asm volatile (
"fclass.d t0, %1\n\t"
"fclass.d t1, %2\n\t"
"li t2, 0x300\n\t"
"and t3, t0, t2\n\t"
"and t4, t1, t2\n\t"
"bnez t3, 1f\n\t"
"bnez t4, 2f\n\t"
"fmin.d %0, %1, %2\n\t"
"j 3f\n\t"
"1:\n\t"
"bnez t4, 4f\n\t"
"fmv.d %0, %2\n\t"
"j 3f\n\t"
"2:\n\t"
"fmv.d %0, %1\n\t"
"j 3f\n\t"
"4:\n\t"
"fmin.d %0, %1, %2\n\t"
"3:\n\t"
: "=f" (res)
: "f" (a), "f" (b)
: "t0", "t1", "t2", "t3", "t4"
);
#else
asm volatile ("fminm.d %0, %1, %2" : "=f" (res) : "f" (a), "f" (b));
#endif
}
read_cycles();
}
read_cycles();
// fli.d
read_cycles();
volatile double res_fli_d[32];
for (int i = 0; i < N; i++) {
res_fli_s[0] = -1.0f;
res_fli_s[1] = -1.0f;
res_fli_s[2] = 0x1p-16f;
res_fli_s[3] = 0x1p-15f;
res_fli_s[4] = 0x1p-14f;
res_fli_s[5] = 0x1p-13f;
res_fli_s[6] = 0x1p-12f;
res_fli_s[7] = 0x1p-11f;
res_fli_s[8] = 0x1p-10f;
res_fli_s[9] = 0x1p-9f;
res_fli_s[10] = 0x1p-8f;
res_fli_s[11] = 0x1p-7f;
res_fli_s[12] = 0x1p-6f;
res_fli_s[13] = 0x1p-5f;
res_fli_s[14] = 0x1p-4f;
res_fli_s[15] = 0x1p-3f;
res_fli_s[16] = 0.25f;
res_fli_s[17] = 0.5f;
res_fli_s[18] = 0.75f;
res_fli_s[19] = 1.0f;
res_fli_s[20] = 1.25f;
res_fli_s[21] = 1.5f;
res_fli_s[22] = 1.75f;
res_fli_s[23] = 2.0f;
res_fli_s[24] = 2.5f;
res_fli_s[25] = 3.0f;
res_fli_s[26] = 4.0f;
res_fli_s[27] = 8.0f;
res_fli_s[28] = 16.0f;
res_fli_s[29] = 32.0f;
res_fli_s[30] = INFINITY;
res_fli_s[31] = NAN;
// -- fli.s --
{
volatile float res[32];
read_cycles();
for (int i = 0; i < N; i++) {
res[0] = -1.0f;
res[1] = -1.0f;
res[2] = 0x1p-16f;
res[3] = 0x1p-15f;
res[4] = 0x1p-14f;
res[5] = 0x1p-13f;
res[6] = 0x1p-12f;
res[7] = 0x1p-11f;
res[8] = 0x1p-10f;
res[9] = 0x1p-9f;
res[10] = 0x1p-8f;
res[11] = 0x1p-7f;
res[12] = 0x1p-6f;
res[13] = 0x1p-5f;
res[14] = 0x1p-4f;
res[15] = 0x1p-3f;
res[16] = 0.25f;
res[17] = 0.5f;
res[18] = 0.75f;
res[19] = 1.0f;
res[20] = 1.25f;
res[21] = 1.5f;
res[22] = 1.75f;
res[23] = 2.0f;
res[24] = 2.5f;
res[25] = 3.0f;
res[26] = 4.0f;
res[27] = 8.0f;
res[28] = 16.0f;
res[29] = 32.0f;
res[30] = INFINITY;
res[31] = NAN;
}
read_cycles();
}
read_cycles();
// fcvtmod.w.d
// -- fli.d --
{
volatile double res[32];
read_cycles();
for (int i = 0; i < N; i++) {
res[0] = -1.0;
res[1] = -1.0;
res[2] = 0x1p-16;
res[3] = 0x1p-15;
res[4] = 0x1p-14;
res[5] = 0x1p-13;
res[6] = 0x1p-12;
res[7] = 0x1p-11;
res[8] = 0x1p-10;
res[9] = 0x1p-9;
res[10] = 0x1p-8;
res[11] = 0x1p-7;
res[12] = 0x1p-6;
res[13] = 0x1p-5;
res[14] = 0x1p-4;
res[15] = 0x1p-3;
res[16] = 0.25;
res[17] = 0.5;
res[18] = 0.75;
res[19] = 1.0;
res[20] = 1.25;
res[21] = 1.5;
res[22] = 1.75;
res[23] = 2.0;
res[24] = 2.5;
res[25] = 3.0;
res[26] = 4.0;
res[27] = 8.0;
res[28] = 16.0;
res[29] = 32.0;
res[30] = INFINITY;
res[31] = NAN;
}
read_cycles();
}
// -- fcvtmod.w.d --
{
static const double inputs[N] = {
2147483649.0, // 2^31 + 1 -> -2147483647 (mod 2^32)
-2147483649.0, // -2^31 - 1 -> 2147483647 (mod 2^32)
4294967297.0, // 2^32 + 1 -> 1 (mod 2^32)
-4294967297.0, // -2^32 - 1 -> -1 (mod 2^32)
3.25, // in-range
0.0, // exact zero
-0.0, // negative zero
INFINITY,
-INFINITY,
NAN
};
volatile double a;
int32_t res;
read_cycles();
for (int i = 0; i < N; i++) {
a = inputs[i];
asm volatile("" : "+f"(a));
#ifndef ZFA
res = emulate_fcvtmod_w_d(a);
#else
asm volatile ("fcvtmod.w.d %0, %1, rtz" : "=r" (res) : "f" (a));
#endif
asm volatile("" : : "r"(res));
}
read_cycles();
}
return 0;
}

BIN
zfhmin.parquet Normal file

Binary file not shown.

View file

@ -1,7 +1,7 @@
#include <math.h>
#include <stdint.h>
#define N 10
#define N 100
static inline uint64_t read_cycles() {
uint64_t start;

View file

@ -2,7 +2,6 @@
#define N 128
#define ITERATIONS 1
// Static "messy" data to ensure the branch predictor cannot "learn" the pattern
static const uint64_t src_a[N] = {
0, 1, 1, 0, 1, 0, 0, 1, 0, 1, 1, 1, 0, 0, 1, 0, 0, 1, 1, 0, 1, 1, 0, 1, 0, 0, 1, 1, 1, 0, 0, 1,
@ -22,6 +21,17 @@ static const uint64_t src_b[N] = {
1130, 1140, 1150, 1160, 1170, 1180, 1190, 1200, 1210, 1220, 1230, 1240, 1250, 1260, 1270, 1280
};
static const uint64_t src_c[N] = {
10, 20, 30, 40, 50, 60, 70, 80, 90, 100, 110, 120, 130, 140, 150, 160,
170, 180, 190, 200, 210, 220, 230, 240, 250, 260, 270, 280, 290, 300, 310, 320,
330, 340, 350, 360, 370, 380, 390, 400, 410, 420, 430, 440, 450, 460, 470, 480,
490, 500, 510, 520, 530, 540, 550, 560, 570, 580, 590, 600, 610, 620, 630, 640,
650, 660, 670, 680, 690, 700, 710, 720, 730, 740, 750, 760, 770, 780, 790, 800,
810, 820, 830, 840, 850, 860, 870, 880, 890, 900, 910, 920, 930, 940, 950, 960,
970, 980, 990, 1000, 1010, 1020, 1030, 1040, 1050, 1060, 1070, 1080, 1090, 1100, 1110, 1120,
1130, 1140, 1150, 1160, 1170, 1180, 1190, 1200, 1210, 1220, 1230, 1240, 1250, 1260, 1270, 1280
};
volatile uint64_t results[N];
static inline uint64_t read_cycles() {
@ -36,8 +46,19 @@ int main() {
for (int i = 0; i < N; i++) {
uint64_t a = src_a[i];
uint64_t b = src_b[i];
uint64_t res;
results[i] = (a != 0) ? b : 0;
#ifdef ZICOND
__asm__ (
"czero.eqz %0, %1, %2"
: "=r" (res) // %0: output
: "r" (b), "r" (a) // %1: rs1, %2: rs2
);
#else
res = (a == 0) ? b : 0;
#endif
results[i] = res;
}
}
read_cycles();
@ -47,10 +68,52 @@ int main() {
for (int i = 0; i < N; i++) {
uint64_t a = src_a[i];
uint64_t b = src_b[i];
uint64_t res;
results[i] = (a != 0) ? b : 0;
#ifdef ZICOND
__asm__ (
"czero.nez %0, %1, %2"
: "=r" (res) // %0: output
: "r" (b), "r" (a) // %1: rs1, %2: rs2
);
#else
res = (a != 0) ? b : 0;
#endif
results[i] = res;
}
}
read_cycles();
read_cycles();
for (int j = 0; j < ITERATIONS; j++) {
for (int i = 0; i < N; i++) {
uint64_t a = src_b[i];
uint64_t b = src_c[i];
uint64_t c = src_a[i];
uint64_t res;
#ifdef ZICOND
uint64_t val_if_true = a | b;
uint64_t val_if_false = a & b;
uint64_t tmp1, tmp2;
__asm__ (
"czero.eqz %0, %3, %2\n\t" // If c != 0, %0 = val_if_true, else 0
"czero.nez %1, %4, %2\n\t" // If c == 0, %1 = val_if_false, else 0
"or %0, %0, %1" // Combine them
: "=&r" (res), "=&r" (tmp2)
: "r" (c), "r" (val_if_true), "r" (val_if_false)
);
#else
// Standard C - likely to compile to a branch if not optimized
res = (c != 0) ? (a | b) : (a & b);
#endif
results[i] = res;
}
}
read_cycles();
return 0;

View file

@ -17,21 +17,17 @@ static inline uint64_t read_cycles() {
}
void benchmark() {
// 1. Widening: _Float16 -> float
uint64_t t0 = read_cycles();
read_cycles();
for (int i = 0; i < N; i++) {
b[i] = (float)a[i];
}
uint64_t t1 = read_cycles();
read_cycles();
// 2. Narrowing: float -> _Float16
uint64_t t2 = read_cycles();
read_cycles();
for (int i = 0; i < N; i++) {
a[i] = (_Float16)b[i];
}
uint64_t t3 = read_cycles();
// In a real app, print (t1-t0) and (t3-t2)
read_cycles();
}
int main() {