Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
60 commits
Select commit Hold shift + click to select a range
2e09d0e
[Autoloop: perf-comparison] Iteration 418: firstValidIndex/lastValidI…
github-actions[bot] Jul 23, 2026
3dc825d
ci: trigger checks
github-actions[bot] Jul 23, 2026
402ebd6
[Autoloop: perf-comparison] Iteration 419: BooleanArray benchmark
github-actions[bot] Jul 24, 2026
e55e590
ci: trigger checks
github-actions[bot] Jul 24, 2026
8d70dac
[Autoloop: perf-comparison] Iteration 420: StringArray benchmark
github-actions[bot] Jul 24, 2026
1b0a1f8
ci: trigger checks
github-actions[bot] Jul 24, 2026
7500ea9
[Autoloop: perf-comparison] Iteration 421: DatetimeArray benchmark
github-actions[bot] Jul 25, 2026
744a1a5
ci: trigger checks
github-actions[bot] Jul 25, 2026
744e7b2
[Autoloop: perf-comparison] Iteration 422: TimedeltaArray benchmark
github-actions[bot] Jul 25, 2026
b1bc5cc
ci: trigger checks
github-actions[bot] Jul 25, 2026
dfda6b1
[Autoloop: perf-comparison] Iteration 423: gaussianKDE benchmark
github-actions[bot] Jul 26, 2026
25e7105
ci: trigger checks
github-actions[bot] Jul 26, 2026
66c5b82
[Autoloop: perf-comparison] Iteration 424: add mode benchmark
github-actions[bot] Jul 26, 2026
f419b5d
ci: trigger checks
github-actions[bot] Jul 26, 2026
6fdb111
[Autoloop: perf-comparison] Iteration 425: add renyiEntropy/tsallisEn…
github-actions[bot] Jul 27, 2026
1a4b041
ci: trigger checks
github-actions[bot] Jul 27, 2026
320e47d
[Autoloop: perf-comparison] Iteration 426: add jointEntropy/condition…
github-actions[bot] Jul 27, 2026
503cb2a
ci: trigger checks
github-actions[bot] Jul 27, 2026
1b98a76
[Autoloop: perf-comparison] Iteration 427: normalizedMI benchmark (4 …
github-actions[bot] Jul 28, 2026
accb8f4
ci: trigger checks
github-actions[bot] Jul 28, 2026
a1c17f7
[Autoloop: perf-comparison] Iteration 428: polyval benchmark
github-actions[bot] Jul 28, 2026
3581d25
ci: trigger checks
github-actions[bot] Jul 28, 2026
fb4cd09
[Autoloop: perf-comparison] Iteration 429: holiday observance benchmarks
github-actions[bot] Jul 29, 2026
9a39d85
ci: trigger checks
github-actions[bot] Jul 29, 2026
5bdd946
[Autoloop: perf-comparison] Iteration 430: information_extended bench…
github-actions[bot] Jul 29, 2026
cf04ae7
ci: trigger checks
github-actions[bot] Jul 29, 2026
c52156c
[Autoloop: perf-comparison] Iteration 431: stata benchmark (readStata…
github-actions[bot] Jul 30, 2026
ea706db
ci: trigger checks
github-actions[bot] Jul 30, 2026
c837e00
[Autoloop: perf-comparison] Iteration 432: SparseArray arithmetic/uti…
github-actions[bot] Jul 30, 2026
95b89db
ci: trigger checks
github-actions[bot] Jul 30, 2026
cc17888
[Autoloop: perf-comparison] Iteration 433: IntegerArray arithmetic op…
github-actions[bot] Jul 31, 2026
8c7373e
ci: trigger checks
github-actions[bot] Jul 31, 2026
842c4ad
[Autoloop: perf-comparison] Iteration 434: applymap benchmark
github-actions[bot] Aug 1, 2026
d9d2911
ci: trigger checks
github-actions[bot] Aug 1, 2026
26400e7
[Autoloop: perf-comparison] Iteration 435: string_accessor benchmark …
github-actions[bot] Aug 1, 2026
77d1ef9
ci: trigger checks
github-actions[bot] Aug 1, 2026
032397e
[Autoloop: perf-comparison] Iteration 436: clip_with_bounds benchmark
github-actions[bot] Aug 2, 2026
44838d0
ci: trigger checks
github-actions[bot] Aug 2, 2026
9fdcff1
[Autoloop: perf-comparison] Iteration 437: swaplevel_dataframe benchmark
github-actions[bot] Aug 2, 2026
5f1924b
ci: trigger checks
github-actions[bot] Aug 2, 2026
646ec8f
[Autoloop: perf-comparison] Iteration 438: information_advanced bench…
github-actions[bot] Aug 3, 2026
513f585
ci: trigger checks
github-actions[bot] Aug 3, 2026
bd73477
[Autoloop: perf-comparison] Iteration 439: format_table benchmark (to…
github-actions[bot] Aug 3, 2026
81285c8
ci: trigger checks
github-actions[bot] Aug 3, 2026
2863f1f
[Autoloop: perf-comparison] Iteration 440: add masked_array and frequ…
github-actions[bot] Aug 4, 2026
71d9caa
ci: trigger checks
github-actions[bot] Aug 4, 2026
1923387
[Autoloop: perf-comparison] Iteration 441: add chi2_contingency and k…
github-actions[bot] Aug 4, 2026
f4032c4
ci: trigger checks
github-actions[bot] Aug 4, 2026
0e5f374
[Autoloop: perf-comparison] Iteration 442: add numeric_extended bench…
github-actions[bot] Aug 5, 2026
4ad24ac
ci: trigger checks
github-actions[bot] Aug 5, 2026
806b80b
[Autoloop: perf-comparison] Iteration 443: add sort_index_columns ben…
github-actions[bot] Aug 5, 2026
60eaabd
ci: trigger checks
github-actions[bot] Aug 5, 2026
3fd006a
[Autoloop: perf-comparison] Iteration 444: bench_datetime_tz (tz_loca…
github-actions[bot] Aug 6, 2026
cf72bc2
ci: trigger checks
github-actions[bot] Aug 6, 2026
8ba6bb0
[Autoloop: perf-comparison] Iteration 445: Add HDF5 round-trip benchm…
github-actions[bot] Aug 6, 2026
ac3f375
ci: trigger checks
github-actions[bot] Aug 6, 2026
aa6a24e
[Autoloop: perf-comparison] Iteration 446: Add Parquet round-trip ben…
github-actions[bot] Aug 7, 2026
1aee12a
ci: trigger checks
github-actions[bot] Aug 7, 2026
5ca3980
[Autoloop: perf-comparison] Iteration 447: Add bench_holiday_calendar…
github-actions[bot] Aug 7, 2026
f083b6e
ci: trigger checks
github-actions[bot] Aug 7, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 40 additions & 0 deletions benchmarks/pandas/bench_applymap.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,40 @@
"""
Benchmark: DataFrame.map (applymap) — element-wise function on every cell.
Mirrors pandas DataFrame.map / DataFrame.applymap.
Dataset: 50,000 rows × 4 columns of float64.
"""
import json
import time
import numpy as np
import pandas as pd

ROWS = 50_000
WARMUP = 5
ITERATIONS = 30

df = pd.DataFrame({
"a": np.arange(ROWS, dtype=np.float64) * 0.5,
"b": np.arange(ROWS, dtype=np.float64) * 1.1,
"c": np.arange(ROWS, dtype=np.float64) * 2.3,
"d": np.arange(ROWS, dtype=np.float64) * 0.7,
})

fn = lambda v: v * 2.0 + 1.0

# pandas >= 2.1 uses df.map(); older versions use df.applymap()
_map = df.map if hasattr(df, "map") else df.applymap

for _ in range(WARMUP):
_map(fn)

start = time.perf_counter()
for _ in range(ITERATIONS):
_map(fn)
total = (time.perf_counter() - start) * 1000

print(json.dumps({
"function": "applymap",
"mean_ms": total / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total,
}))
43 changes: 43 additions & 0 deletions benchmarks/pandas/bench_boolean_array.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,43 @@
"""Benchmark: BooleanArray — nullable boolean extension array operations.
N=100_000 elements with ~10% nulls using pandas BooleanArray.
Tests: array creation, any, all, sum, and, or, invert, fillna.
"""
import json
import time
import pandas as pd

N = 100_000
WARMUP = 5
ITERATIONS = 50

# Same pattern as TS version (~10% nulls)
raw = [(None if i % 10 == 0 else bool(i % 3 != 0)) for i in range(N)]
raw2 = [(None if i % 7 == 0 else bool(i % 2 == 0)) for i in range(N)]


def run():
a = pd.array(raw, dtype="boolean")
b = pd.array(raw2, dtype="boolean")
_ = a.any(skipna=True)
_ = a.all(skipna=True)
_ = a.sum(skipna=True)
_ = a & b
_ = a | b
_ = ~a
_ = a.fillna(False)


for _ in range(WARMUP):
run()

start = time.perf_counter()
for _ in range(ITERATIONS):
run()
total = (time.perf_counter() - start) * 1000

print(json.dumps({
"function": "boolean_array",
"mean_ms": total / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total,
}))
47 changes: 47 additions & 0 deletions benchmarks/pandas/bench_chi2_contingency.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
"""
Benchmark: chi2Contingency — chi-squared test of independence on contingency tables.
Mirrors tsb bench_chi2_contingency.ts.
Dataset: 500 iterations over 4×4, 5×3, and 3×5 contingency tables.
Outputs JSON: {"function": "chi2_contingency", "mean_ms": ..., "iterations": ..., "total_ms": ...}
"""
import json
import time
import numpy as np
from scipy.stats import chi2_contingency

WARMUP = 10
ITERATIONS = 500

table4x4 = np.array([
[10, 20, 30, 15],
[25, 35, 10, 20],
[15, 10, 25, 30],
[20, 15, 35, 10],
], dtype=float)
table5x3 = np.array([
[50, 30, 20],
[40, 45, 15],
[35, 25, 40],
[20, 50, 30],
[55, 10, 35],
], dtype=float)
table3x5 = np.array([
[10, 20, 15, 25, 30],
[30, 15, 25, 10, 20],
[20, 30, 10, 35, 5],
], dtype=float)

for _ in range(WARMUP):
chi2_contingency(table4x4)
chi2_contingency(table5x3)
chi2_contingency(table3x5)

t0 = time.perf_counter()
for _ in range(ITERATIONS):
chi2_contingency(table4x4)
chi2_contingency(table5x3)
chi2_contingency(table3x5)
total_ms = (time.perf_counter() - t0) * 1000
mean_ms = total_ms / ITERATIONS

print(json.dumps({"function": "chi2_contingency", "mean_ms": mean_ms, "iterations": ITERATIONS, "total_ms": total_ms}))
36 changes: 36 additions & 0 deletions benchmarks/pandas/bench_clip_with_bounds.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
"""
Benchmark: pandas Series.clip / DataFrame.clip with array/Series bounds.
Outputs JSON: {"function": "clip_with_bounds", "mean_ms": ..., "iterations": ..., "total_ms": ...}
"""
import json
import time
import numpy as np
import pandas as pd

ROWS = 100_000
WARMUP = 5
ITERATIONS = 20

data = np.array([(i % 200) - 100 for i in range(ROWS)], dtype=float)
lower_arr = np.full(ROWS, -30.0)
upper_arr = np.full(ROWS, 30.0)
s = pd.Series(data)
lower_s = pd.Series(lower_arr)
upper_s = pd.Series(upper_arr)

df_cols = {f"col{c}": [(i + c * 10) % 200 - 100 for i in range(ROWS)] for c in range(4)}
df = pd.DataFrame(df_cols, dtype=float)
df_lower = pd.Series(np.full(ROWS, -30.0))
df_upper = pd.Series(np.full(ROWS, 30.0))

for _ in range(WARMUP):
s.clip(lower=lower_s, upper=upper_s)
df.clip(lower=df_lower, upper=df_upper, axis=0)

start = time.perf_counter()
for _ in range(ITERATIONS):
s.clip(lower=lower_s, upper=upper_s)
df.clip(lower=df_lower, upper=df_upper, axis=0)
total = (time.perf_counter() - start) * 1000

print(json.dumps({"function": "clip_with_bounds", "mean_ms": total / ITERATIONS, "iterations": ITERATIONS, "total_ms": total}))
41 changes: 41 additions & 0 deletions benchmarks/pandas/bench_datetime_array.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,41 @@
"""Benchmark: DatetimeArray — nullable datetime extension array operations.
N=100_000 elements with ~10% nulls using pandas DatetimeArray.
Tests: from_sequence, year, month, day, isna, notna, fillna.
"""
import json
import time
import pandas as pd
import numpy as np

N = 100_000
WARMUP = 3
ITERATIONS = 50

base = pd.Timestamp("2020-01-01")
raw = [(None if i % 10 == 0 else base + pd.Timedelta(days=i)) for i in range(N)]


def run():
a = pd.array(raw, dtype="datetime64[ns]")
_ = a.year
_ = a.month
_ = a.day
_ = pd.isna(a)
_ = ~pd.isna(a)
_ = a.fillna(pd.Timestamp("2000-01-01"))


for _ in range(WARMUP):
run()

start = time.perf_counter()
for _ in range(ITERATIONS):
run()
total = (time.perf_counter() - start) * 1000

print(json.dumps({
"function": "datetime_array",
"mean_ms": total / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total,
}))
30 changes: 30 additions & 0 deletions benchmarks/pandas/bench_datetime_tz.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
"""Benchmark: tz_localize and tz_convert on DatetimeIndex (pandas equivalent)."""
import json
import time
import pandas as pd

SIZE = 10_000
WARMUP = 5
ITERATIONS = 50

naive = pd.date_range(start="2024-01-01", periods=SIZE, freq="h")

# Warm-up
for _ in range(WARMUP):
ny = naive.tz_localize("America/New_York")
ny.tz_convert("UTC")
ny.tz_convert("Europe/London")

start = time.perf_counter()
for _ in range(ITERATIONS):
ny = naive.tz_localize("America/New_York")
ny.tz_convert("UTC")
ny.tz_convert("Europe/London")
total = (time.perf_counter() - start) * 1000

print(json.dumps({
"function": "datetime_tz",
"mean_ms": total / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total,
}))
45 changes: 45 additions & 0 deletions benchmarks/pandas/bench_first_last_valid_index.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
"""
Benchmark: first_valid_index / last_valid_index
Outputs JSON: {"function": "first_last_valid_index", "mean_ms": ..., "iterations": ..., "total_ms": ...}
"""
import json
import time
import numpy as np
import pandas as pd

N = 100_000

# Series where first valid is near the start (a few NaN at beginning)
data_start = np.where(np.arange(N) < 10, np.nan, np.arange(N, dtype=float))
series_start = pd.Series(data_start)

# Series where last valid is near the end (a few NaN at the end)
data_end = np.where(np.arange(N) >= N - 10, np.nan, np.arange(N, dtype=float))
series_end = pd.Series(data_end)

# Series with NaN scattered throughout
data_mixed = np.where(np.arange(N) % 7 == 0, np.nan, np.arange(N, dtype=float))
series_mixed = pd.Series(data_mixed)

# Warm-up
for _ in range(20):
series_start.first_valid_index()
series_end.last_valid_index()
series_mixed.first_valid_index()
series_mixed.last_valid_index()

iterations = 500
start = time.perf_counter()
for _ in range(iterations):
series_start.first_valid_index()
series_end.last_valid_index()
series_mixed.first_valid_index()
series_mixed.last_valid_index()
total_ms = (time.perf_counter() - start) * 1000

print(json.dumps({
"function": "first_last_valid_index",
"mean_ms": total_ms / iterations,
"iterations": iterations,
"total_ms": total_ms,
}))
39 changes: 39 additions & 0 deletions benchmarks/pandas/bench_format_table.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,39 @@
"""Benchmark: to_markdown / to_latex / Series.to_markdown on a 1000-row DataFrame"""
import json, time, subprocess, sys
try:
import tabulate # noqa: F401
except ImportError:
subprocess.run([sys.executable, "-m", "pip", "install", "tabulate", "--quiet"], check=False)
import numpy as np
import pandas as pd

ROWS = 1_000
WARMUP = 3
ITERATIONS = 20

data = {
"a": np.arange(ROWS) * 1.1,
"b": np.sin(np.arange(ROWS)) * 100,
"c": np.arange(ROWS) % 7,
}
df = pd.DataFrame(data)
s = pd.Series(np.arange(ROWS) * 2.5, name="x")

for _ in range(WARMUP):
df.to_markdown()
df.to_latex()
s.to_markdown()

start = time.perf_counter()
for _ in range(ITERATIONS):
df.to_markdown()
df.to_latex()
s.to_markdown()
total = (time.perf_counter() - start) * 1000

print(json.dumps({
"function": "format_table",
"mean_ms": total / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total,
}))
36 changes: 36 additions & 0 deletions benchmarks/pandas/bench_frequencies.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
"""Benchmark: frequencies — to_offset and infer_freq.
Tests parsing frequency strings and inferring frequency from date arrays.
"""
import json
import time
import pandas as pd

WARMUP = 5
ITERATIONS = 50

FREQ_STRINGS = ["D", "h", "min", "s", "ME", "MS", "YE", "YS", "W", "3ME", "2h", "QE", "QS"]

# Build a regularly-spaced daily date index for infer_freq
daily_index = pd.date_range("2020-01-01", periods=365, freq="D")


def run():
for freq in FREQ_STRINGS:
pd.tseries.frequencies.to_offset(freq)
pd.infer_freq(daily_index)


for _ in range(WARMUP):
run()

start = time.perf_counter()
for _ in range(ITERATIONS):
run()
total = (time.perf_counter() - start) * 1000

print(json.dumps({
"function": "frequencies",
"mean_ms": total / ITERATIONS,
"iterations": ITERATIONS,
"total_ms": total,
}))
Loading
Loading