import plotly.express as px
import pandas as pd
# df = pd.read_parquet("era5-pds/measurements-m1.parquet")
# df = pd.read_parquet("era5-pds/measurements-i10k.parquet")
# df = pd.read_parquet("era5-pds/measurements-ryzen3.parquet")
# df = pd.read_parquet("era5-pds/measurements-i13k.parquet")
df = pd.read_parquet("era5-pds/measurements-always-split-clang.parquet")
df = df.query("clevel > 0") # get rid of no compression results
category_orders = {"dset": ["flux", "wind", "pressure", "precip", "snow"],
"filter": ["nofilter", "shuffle", "bitshuffle", "bytedelta"]}
labels = {
"cratio": "Compression ratio (x times)",
"cspeed": "Compression speed (GB/s)",
"dspeed": "Decompression speed (GB/s)",
"codec": "Codec",
"dset": "Dataset",
"filter": "Filter",
"cratio * cspeed": "Compression ratio x Compression speed",
"cratio * dspeed": "Compression ratio x Decompression speed",
"cratio * cspeed * dspeed": "Compression ratio x Compression x Decompression speeds",
}
hover_data = {"filter": False, "codec": True, "cratio": ':.1f', "cspeed": ':.2f',
"dspeed": ':.2f', "dset": True, "clevel": True}
fig = px.box(df, x="cratio", color="filter", points="all", hover_data=hover_data,
labels=labels, range_x=(0, 60), range_y=(-.4, .35),)
fig.update_layout(
title={
'text': "Compression ratio vs filter (larger is better)",
#'y':0.9,
'x':0.25,
'xanchor': 'left',
#'yanchor': 'top'
},
#xaxis_title="Filter",
)
fig.show()
hover_data = {"filter": False, "codec": True, "cratio": ':.1f', "cspeed": ':.2f', "dspeed": ':.2f',
"dset": False, "clevel": True}
fig = px.strip(df, y="cratio", x="dset", color="filter", hover_data=hover_data, labels=labels,
category_orders=category_orders)
fig.show()
hover_data = {"filter": False, "codec": False, "cratio": ':.1f', "cspeed": ':.2f', "dspeed": ':.2f',
"dset": True, "clevel": True}
fig = px.strip(df, y="cratio", x="codec", color="filter", labels=labels, hover_data=hover_data)
fig.show()
df["cratio * cspeed"] = df["cratio"] * df["cspeed"]
df["cratio * dspeed"] = df["cratio"] * df["dspeed"]
df["cratio * cspeed * dspeed"] = df["cratio"] * df["cspeed"] * df["dspeed"]
df_mean = df.groupby(['filter', 'clevel', 'codec']).mean(numeric_only=True).reset_index(level=[0,1,2])
df_mean2 = df.groupby(['filter', 'dset']).mean(numeric_only=True).reset_index(level=[0,1])
df_mean
filter | clevel | codec | cspeed | dspeed | cratio | cratio * cspeed | cratio * dspeed | cratio * cspeed * dspeed | |
---|---|---|---|---|---|---|---|---|---|
0 | bitshuffle | 1 | BLOSCLZ | 8.158378 | 61.902774 | 8.818274 | 95.872569 | 615.330844 | 6861.726441 |
1 | bitshuffle | 1 | LZ4 | 8.988032 | 70.387148 | 11.901284 | 119.658314 | 908.784873 | 9239.900449 |
2 | bitshuffle | 1 | LZ4HC | 6.441387 | 69.976505 | 13.051075 | 119.072954 | 1003.947510 | 9418.486582 |
3 | bitshuffle | 1 | ZLIB | 6.769396 | 28.916222 | 11.757224 | 100.259251 | 428.335272 | 3836.397273 |
4 | bitshuffle | 1 | ZSTD | 9.408273 | 45.534939 | 16.468393 | 212.190890 | 917.739040 | 12367.013862 |
... | ... | ... | ... | ... | ... | ... | ... | ... | ... |
75 | shuffle | 9 | BLOSCLZ | 10.010134 | 66.752577 | 11.330588 | 174.955069 | 964.198731 | 16355.239339 |
76 | shuffle | 9 | LZ4 | 10.945267 | 84.148562 | 11.119430 | 172.724473 | 1153.930180 | 19146.833696 |
77 | shuffle | 9 | LZ4HC | 1.941989 | 95.847395 | 14.480430 | 56.657586 | 1645.758874 | 7118.267050 |
78 | shuffle | 9 | ZLIB | 0.947060 | 23.953246 | 17.874545 | 29.312204 | 577.526668 | 1091.405870 |
79 | shuffle | 9 | ZSTD | 0.135023 | 52.805664 | 19.741196 | 3.460495 | 1449.888643 | 272.287869 |
80 rows × 9 columns
fig = px.bar(df_mean, y="cratio", x="codec", color="filter", category_orders=category_orders,
barmode="group", facet_col="clevel", labels=labels, title="Compression ratio (mean)")
fig.show()
fig = px.bar(df_mean, y="cspeed", x="codec", color="filter", category_orders=category_orders,
barmode="group", facet_col="clevel", labels=labels, title="Compression speed (mean)")
fig.show()
fig = px.bar(df_mean2, y="cspeed", x="filter", facet_col="dset", color="filter", log_y=True,
labels=labels, category_orders=category_orders)
fig.show()
fig = px.strip(df, y="cspeed", x="codec", color="filter", hover_data=hover_data, labels=labels)
fig.show()
fig = px.bar(df_mean, y="dspeed", x="codec", color="filter",
category_orders=category_orders, barmode="group",
facet_col="clevel", labels=labels, title="Decompression speed (mean)")
fig.show()
fig = px.bar(df_mean2, y="dspeed", x="filter", facet_col="dset", color="filter", log_y=True,
labels=labels, category_orders=category_orders)
fig.show()
fig = px.strip(df, y="dspeed", x="codec", color="filter", hover_data=hover_data, labels=labels)
fig.show()
hover_data = {"filter": True, "codec": True, "cratio": ':.1f', "cspeed": ':.2f',
"dspeed": ':.2f', "dset": True, "clevel": True}
fig = px.scatter(df, y="cratio", x="cspeed", color="filter", log_y=True,
hover_data=hover_data, labels=labels)
fig.show()
fig = px.box(df, y="cratio * cspeed", x="codec", color="filter", log_y=True,
hover_data=hover_data, labels=labels)
fig.show()
fig = px.bar(df_mean, y="cratio * cspeed", x="codec", color="filter", log_y=True,
labels=labels, facet_col="clevel", barmode="group", category_orders=category_orders)
fig.show()
fig = px.bar(df_mean2, y="cratio * cspeed", x="filter", facet_col="dset", color="filter", log_y=True,
labels=labels, category_orders=category_orders)
fig.show()
hover_data = {"filter": True, "codec": True, "cratio": ':.1f', "cspeed": ':.2f',
"dspeed": ':.2f', "dset": True, "clevel": True}
fig = px.scatter(df, y="cratio", x="dspeed", color="filter", log_y=True,
hover_data=hover_data, labels=labels)
fig.show()
fig = px.box(df, y="cratio * dspeed", x="codec", color="filter", log_y=True,
hover_data=hover_data, labels=labels, category_orders=category_orders)
fig.show()
fig = px.bar(df_mean, y="cratio * dspeed", x="codec", color="filter", log_y=True,
labels=labels, facet_col="clevel", barmode="group", category_orders=category_orders)
fig.show()
fig = px.bar(df_mean2, y="cratio * dspeed", x="filter", facet_col="dset", color="filter", log_y=True,
labels=labels, category_orders=category_orders)
fig.show()
fig = px.box(df, y="cratio * cspeed * dspeed", x="codec", color="filter",
log_y=True, hover_data=hover_data, labels=labels, category_orders=category_orders)
fig.show()
fig = px.bar(df_mean, y="cratio * cspeed * dspeed", x="codec", color="filter", log_y=True,
labels=labels, facet_col="clevel", barmode="group", category_orders=category_orders)
fig.show()
fig = px.bar(df_mean2, y="cratio * cspeed * dspeed", x="filter", facet_col="dset", color="filter", log_y=True,
labels=labels, category_orders=category_orders)
fig.show()