Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
18 commits
Select commit Hold shift + click to select a range
4397e30
[Autoloop: perf-comparison] Iteration 482: add wasm_agg_ops benchmark…
github-actions[bot] Aug 24, 2026
bb3aa91
ci: trigger checks
github-actions[bot] Aug 24, 2026
1197660
[Autoloop: perf-comparison] Iteration 483: add wasm_rolling_stats ben…
github-actions[bot] Aug 25, 2026
410d3b8
ci: trigger checks
github-actions[bot] Aug 25, 2026
6808d33
[Autoloop: perf-comparison] Iteration 484: add to_dict_series_orient …
github-actions[bot] Aug 25, 2026
25ce6e8
ci: trigger checks
github-actions[bot] Aug 25, 2026
c090288
[Autoloop: perf-comparison] Iteration 485: add registerOption benchma…
github-actions[bot] Aug 26, 2026
45cb293
ci: trigger checks
github-actions[bot] Aug 26, 2026
78ffce6
perf: add MultiIndex.toList() benchmark pair
github-actions[bot] Aug 27, 2026
5090240
ci: trigger checks
github-actions[bot] Aug 27, 2026
fceac8c
[Autoloop: perf-comparison] Iteration 487: add string_array_str_ops b…
github-actions[bot] Aug 27, 2026
011cb3b
ci: trigger checks
github-actions[bot] Aug 27, 2026
56519d0
[Autoloop: perf-comparison] Iteration 488: Add ewm benchmark
github-actions[bot] Aug 28, 2026
49578de
ci: trigger checks
github-actions[bot] Aug 28, 2026
85dc945
[Autoloop: perf-comparison] Iteration 489: add string_array_cat bench…
github-actions[bot] Aug 28, 2026
c1f5164
ci: trigger checks
github-actions[bot] Aug 28, 2026
fd0704a
[Autoloop: perf-comparison] Iteration 490: add series_rename_ops benc…
github-actions[bot] Aug 29, 2026
3d3bdcb
ci: trigger checks
github-actions[bot] Aug 29, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
23 changes: 23 additions & 0 deletions benchmarks/pandas/bench_ewm.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
"""Benchmark: ewm (Exponentially Weighted Moving) aggregations on 100k-element pandas Series"""
import json, time, math
import numpy as np
import pandas as pd

ROWS = 100_000
WARMUP = 3
ITERATIONS = 10
data = [math.sin(i * 0.01) * 100 + 50 for i in range(ROWS)]
s = pd.Series(data)

for _ in range(WARMUP):
s.ewm(span=20).mean()
s.ewm(span=20).std()
s.ewm(span=20).var()

start = time.perf_counter()
for _ in range(ITERATIONS):
s.ewm(span=20).mean()
s.ewm(span=20).std()
s.ewm(span=20).var()
total = (time.perf_counter() - start) * 1000
print(json.dumps({"function": "ewm", "mean_ms": total / ITERATIONS, "iterations": ITERATIONS, "total_ms": total}))
20 changes: 20 additions & 0 deletions benchmarks/pandas/bench_multi_index_to_list.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
"""Benchmark: MultiIndex.tolist() on 100k-pair MultiIndex"""
import json, time
import pandas as pd

ROWS = 100_000
WARMUP = 3
ITERATIONS = 10
a = [f"a{i % 100}" for i in range(ROWS)]
b = [i % 1000 for i in range(ROWS)]
tuples = list(zip(a, b))
mi = pd.MultiIndex.from_tuples(tuples)

for _ in range(WARMUP):
mi.tolist()

start = time.perf_counter()
for _ in range(ITERATIONS):
mi.tolist()
total = (time.perf_counter() - start) * 1000
print(json.dumps({"function": "multi_index_to_list", "mean_ms": total / ITERATIONS, "iterations": ITERATIONS, "total_ms": total}))
79 changes: 79 additions & 0 deletions benchmarks/pandas/bench_register_option.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
"""
Benchmark: register_option — register custom options with pandas' options system.

Mirrors tsb registerOption which wraps pandas' core config register_option API.
Uses pandas.core.config_init / _config._registered_options to register custom
options with defaults and validators.

Outputs JSON: {"function": "register_option", "mean_ms": ..., "iterations": ..., "total_ms": ...}
"""
import json
import time

import pandas as pd

WARMUP = 5
ITERATIONS = 1_000

key_counter = [0]


def register_and_exercise():
key = f"bench.custom_{key_counter[0]}"
key_counter[0] += 1
# pandas does not expose a public register_option in the top-level namespace,
# but it is accessible via pd.core.config.register_option (internal API).
# We simulate the equivalent pattern: register → get → set → reset.
try:
pd.core.config.register_option(key, 42, "A custom numeric option for benchmarking.")
except Exception:
pass # already registered or unavailable
try:
v = pd.get_option(key)
pd.set_option(key, 99)
pd.reset_option(key)
_ = v
except Exception:
pass


def register_with_validator():
key = f"bench.validated_{key_counter[0]}"
key_counter[0] += 1

def validator(val):
if not isinstance(val, (int, float)) or val < 0:
raise ValueError("must be a non-negative number")

try:
pd.core.config.register_option(key, 10, "A validated option.", validator=validator)
except Exception:
pass
try:
pd.set_option(key, 50)
pd.reset_option(key)
except Exception:
pass


# Warm-up
for _ in range(WARMUP):
register_and_exercise()
register_with_validator()

start = time.perf_counter()
for _ in range(ITERATIONS):
register_and_exercise()
register_with_validator()
total_ms = (time.perf_counter() - start) * 1000

print(
json.dumps(
{
"function": "register_option",
"mean_ms": total_ms / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total_ms,
}
)
)
49 changes: 49 additions & 0 deletions benchmarks/pandas/bench_series_rename_ops.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,49 @@
"""
Benchmark: Series.add_prefix / add_suffix / set_axis / DataFrame.set_axis / Series.to_frame
— matching pandas equivalent for bench_series_rename_ops.ts.

Mirrors tsb addPrefixSeries, addSuffixSeries, setAxisSeries, setAxisDataFrame, seriesToFrame.

Outputs JSON: {"function": "series_rename_ops", "mean_ms": ..., "iterations": ..., "total_ms": ...}
"""
import json
import time
import pandas as pd
import numpy as np

SIZE = 100_000
WARMUP = 5
ITERATIONS = 30

data = [i * 0.5 for i in range(SIZE)]
labels = [f"row_{i}" for i in range(SIZE)]
new_labels = [f"new_{i}" for i in range(SIZE)]

s = pd.Series(data, index=labels, name="values")
df = pd.DataFrame({"a": data, "b": [-v for v in data]}, index=labels)

for _ in range(WARMUP):
s.add_prefix("pre_")
s.add_suffix("_suf")
s.set_axis(new_labels)
df.set_axis(new_labels)
s.to_frame()
s.to_frame(name="renamed")

start = time.perf_counter()
for _ in range(ITERATIONS):
s.add_prefix("pre_")
s.add_suffix("_suf")
s.set_axis(new_labels)
df.set_axis(new_labels)
s.to_frame()
s.to_frame(name="renamed")
total = time.perf_counter() - start

total_ms = total * 1000
print(json.dumps({
"function": "series_rename_ops",
"mean_ms": total_ms / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total_ms,
}))
49 changes: 49 additions & 0 deletions benchmarks/pandas/bench_string_array_cat.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,49 @@
"""
Benchmark: Series.str.cat() — element-wise string concatenation with separator.
N=100_000 nullable strings (~10% nulls) in each of two string Series;
str.cat() joins them pairwise with a separator character.

Mirrors bench_string_array_cat.ts (tsb StringArray.cat("-", other)).

Outputs JSON: {"function": "string_array_cat", "mean_ms": ..., "iterations": ..., "total_ms": ...}
"""

import json
import time

import pandas as pd
import numpy as np

N = 100_000
WARMUP = 3
ITERATIONS = 50

WORDS_A = ["hello", "world", "foo", "bar", "baz", "qux", "quux", "corge", "grault", "garply"]
WORDS_B = ["alpha", "beta", "gamma", "delta", "epsilon", "zeta", "eta", "theta", "iota", "kappa"]

raw_a = [None if i % 10 == 0 else WORDS_A[i % len(WORDS_A)] for i in range(N)]
raw_b = [None if i % 7 == 0 else WORDS_B[i % len(WORDS_B)] for i in range(N)]

s_a = pd.Series(raw_a, dtype="string")
s_b = pd.Series(raw_b, dtype="string")


def run() -> None:
s_a.str.cat(s_b, sep="-", na_rep=None)


for _ in range(WARMUP):
run()

t0 = time.perf_counter()
for _ in range(ITERATIONS):
run()
total_s = time.perf_counter() - t0
total_ms = total_s * 1000

print(json.dumps({
"function": "string_array_cat",
"mean_ms": total_ms / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total_ms,
}))
47 changes: 47 additions & 0 deletions benchmarks/pandas/bench_string_array_str_ops.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
"""
Benchmark: StringArray additional string operations —
lstrip, rstrip, startswith, endswith, replace, zfill
on a 100k-element nullable StringDtype array (~10 % nulls).

Mirrors pandas pd.array([...], dtype="string") str methods:
str.lstrip, str.rstrip, str.startswith, str.endswith, str.replace, str.zfill

Outputs JSON: {"function": "string_array_str_ops", "mean_ms": ..., "iterations": ..., "total_ms": ...}
"""
import json
import time
import pandas as pd

N = 100_000
WARMUP = 3
ITERATIONS = 50

WORDS = [" hello world ", " foo bar ", "baz qux ", " quux", "corge", "grault ", "garply"]
raw = [None if i % 10 == 0 else WORDS[i % len(WORDS)] for i in range(N)]

a = pd.array(raw, dtype="string")


def run() -> None:
a.str.lstrip()
a.str.rstrip()
a.str.startswith(" he")
a.str.endswith("ld ")
a.str.replace("hello", "hi", regex=False)
a.str.zfill(12)


for _ in range(WARMUP):
run()

start = time.perf_counter()
for _ in range(ITERATIONS):
run()
total_ms = (time.perf_counter() - start) * 1000

print(json.dumps({
"function": "string_array_str_ops",
"mean_ms": total_ms / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total_ms,
}))
38 changes: 38 additions & 0 deletions benchmarks/pandas/bench_to_dict_series_orient.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,38 @@
"""
Benchmark: DataFrame.to_dict(orient="series") — converts each column to a pandas Series.

Mirrors tsb toDictOriented(df, "series").

Outputs JSON: {"function": "to_dict_series_orient", "mean_ms": ..., "iterations": ..., "total_ms": ...}
"""
import json
import time
import numpy as np
import pandas as pd

ROWS = 10_000
WARMUP = 5
ITERATIONS = 30

df = pd.DataFrame({
"id": np.arange(ROWS),
"value": np.arange(ROWS) * 1.5,
"label": [f"item_{i % 100}" for i in range(ROWS)],
"score": np.sin(np.arange(ROWS) * 0.01) * 100,
"flag": np.arange(ROWS) % 2 == 0,
})

for _ in range(WARMUP):
df.to_dict(orient="series")

t0 = time.perf_counter()
for _ in range(ITERATIONS):
df.to_dict(orient="series")
total = (time.perf_counter() - t0) * 1000

print(json.dumps({
"function": "to_dict_series_orient",
"mean_ms": total / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total,
}))
51 changes: 51 additions & 0 deletions benchmarks/pandas/bench_wasm_agg_ops.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
"""
Benchmark: numpy aggregate operations — np.sum, np.mean, np.min, np.max, np.var, np.std, np.median
plus pandas rolling and expanding window ops on a 100k-element float64 array.

Mirrors tsb bench_wasm_agg_ops.ts.

Outputs JSON: {"function": "wasm_agg_ops", "mean_ms": ..., "iterations": ..., "total_ms": ...}
"""
import json
import time
import numpy as np
import pandas as pd

SIZE = 100_000
WINDOW = 50
MIN_PERIODS = 1
WARMUP = 3
ITERATIONS = 20

data = np.sin(np.arange(SIZE) * 0.001) * 1000
series = pd.Series(data)


def run():
np.sum(data)
np.mean(data)
np.min(data)
np.max(data)
np.var(data, ddof=1)
np.std(data, ddof=1)
np.median(data)
series.rolling(window=WINDOW, min_periods=MIN_PERIODS).sum()
series.rolling(window=WINDOW, min_periods=MIN_PERIODS).mean()
series.expanding(min_periods=MIN_PERIODS).sum()
series.expanding(min_periods=MIN_PERIODS).mean()


for _ in range(WARMUP):
run()

start = time.perf_counter()
for _ in range(ITERATIONS):
run()
total = (time.perf_counter() - start) * 1000 # ms

print(json.dumps({
"function": "wasm_agg_ops",
"mean_ms": total / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total,
}))
Loading
Loading