Skip to content

Commit f6697d9

Browse files
authored
feat(site): add benches snapshot page (#1789)
## What Adds the `/benches` site page as a latest benchmark/eval snapshot, with links to the latest Markdown reports and result folders. ## Why Bench, Criterion, and eval artifacts were hard to inspect together. The site now gives a simple current-state view while keeping raw reports one click away. ## How - Aggregates saved benchmark, Criterion, and eval artifacts with `site/scripts/build-performance-data.mjs` - Renders `/benches` from generated `performance-timeline.json` - Adds result-folder/report links, category descriptions, latest-run values, and clearer eval pass-rate labels - Adds `specs/performance-results.md` and updates run recipes so saved bench/eval runs refresh the site data - Fixes `bashkit-bench --save path/to/file.json` so custom output directories are preserved ## Risk - Low - Site data could become stale if result schemas change without updating the transformer; the new spec calls this out and `prebuild` regenerates the data. ## Checklist - [x] Tests added or updated - [x] Backward compatibility considered - [x] `just pre-pr` - [x] `pnpm run build` - [x] Browser smoke test for `/benches`
1 parent d04b4f5 commit f6697d9

14 files changed

Lines changed: 5979 additions & 35 deletions

File tree

‎AGENTS.md‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -48,6 +48,7 @@ Fix root cause. Unsure: read more code; if stuck, ask w/ short options. Unrecogn
4848
| sqlite-builtin | Embedded SQLite via Turso (MemoryIO + VfsIO backends, dot-commands) |
4949
| coreutils-args-port | Port uutils `uu_app()` clap definitions (args mode) and platform-clean uucore modules (module mode, manifest-driven) into bashkit via codegen |
5050
| credential-injection | Transparent per-host credential injection for outbound HTTP requests, without exposing secrets to sandboxed scripts |
51+
| performance-results | Benchmark/eval result locations and `/benches` site aggregation contract |
5152

5253
### Documentation
5354

‎crates/bashkit-bench/README.md‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -127,6 +127,10 @@ cargo run -p bashkit-bench --release -- --list
127127
| `--verbose` | Show per-benchmark timing details |
128128
| `--list` | List available benchmarks |
129129

130+
Saved JSON/Markdown reports in `crates/bashkit-bench/results/` feed the site
131+
`/benches` page. See `specs/performance-results.md` for the aggregation
132+
contract.
133+
130134
## Prerequisites
131135

132136
| Runner | Setup |

‎crates/bashkit-bench/src/main.rs‎

Lines changed: 52 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -391,21 +391,15 @@ async fn main() -> Result<()> {
391391

392392
// Save if requested
393393
if let Some(ref save_arg) = args.save {
394-
let base_name = if save_arg.is_empty() {
395-
// Auto-generate filename with moniker and timestamp
396-
let timestamp = chrono_lite_now();
397-
format!("bench-{}-{}", system_info.moniker, timestamp)
398-
} else {
399-
// Use provided name, strip extension if present
400-
let path = PathBuf::from(save_arg);
401-
path.file_stem()
402-
.and_then(|s| s.to_str())
403-
.unwrap_or("bench-results")
404-
.to_string()
405-
};
394+
let timestamp = chrono_lite_now();
395+
let base_path = save_base_path(save_arg, &system_info.moniker, &timestamp);
406396

407-
let json_path = format!("{}.json", base_name);
408-
let md_path = format!("{}.md", base_name);
397+
let json_path = base_path.with_extension("json");
398+
let md_path = base_path.with_extension("md");
399+
400+
if let Some(parent) = json_path.parent() {
401+
std::fs::create_dir_all(parent).context("Failed to create results directory")?;
402+
}
409403

410404
// Save JSON
411405
let json = serde_json::to_string_pretty(&report)?;
@@ -418,14 +412,30 @@ async fn main() -> Result<()> {
418412
println!(
419413
"\n{} results to:\n - {}\n - {}",
420414
"Saved".green(),
421-
json_path,
422-
md_path
415+
json_path.display(),
416+
md_path.display()
423417
);
424418
}
425419

426420
Ok(())
427421
}
428422

423+
fn save_base_path(save_arg: &str, moniker: &str, timestamp: &str) -> PathBuf {
424+
if save_arg.is_empty() {
425+
// Auto-generate inside the repo-tracked results folder so site builds
426+
// can pick up fresh benchmark runs.
427+
return PathBuf::from("crates/bashkit-bench/results")
428+
.join(format!("bench-{}-{}", moniker, timestamp));
429+
}
430+
431+
let path = PathBuf::from(save_arg);
432+
if path.extension().is_some() {
433+
path.with_extension("")
434+
} else {
435+
path
436+
}
437+
}
438+
429439
async fn run_benchmark(
430440
runner: &mut Runner,
431441
case: &BenchCase,
@@ -780,3 +790,29 @@ fn print_summary(summary: &BenchSummary) {
780790
println!();
781791
}
782792
}
793+
794+
#[cfg(test)]
795+
mod tests {
796+
use super::save_base_path;
797+
use std::path::PathBuf;
798+
799+
#[test]
800+
fn save_base_path_defaults_to_site_indexed_results_dir() {
801+
assert_eq!(
802+
save_base_path("", "vm-linux-x86_64", "1779764460"),
803+
PathBuf::from("crates/bashkit-bench/results/bench-vm-linux-x86_64-1779764460")
804+
);
805+
}
806+
807+
#[test]
808+
fn save_base_path_preserves_custom_directory_and_strips_extension() {
809+
assert_eq!(
810+
save_base_path(
811+
"crates/bashkit-bench/results/manual-test.json",
812+
"ignored",
813+
"ignored"
814+
),
815+
PathBuf::from("crates/bashkit-bench/results/manual-test")
816+
);
817+
}
818+
}

‎justfile‎

Lines changed: 29 additions & 17 deletions
Original file line numberDiff line numberDiff line change
@@ -124,63 +124,75 @@ run-script file:
124124

125125
# === Benchmarks ===
126126

127-
# Run benchmarks comparing bashkit to bash
127+
# Run benchmarks comparing bashkit to bash and save site-indexed JSON/Markdown results
128128
bench:
129-
cargo run -p bashkit-bench --release
129+
cargo run -p bashkit-bench --release -- --save
130+
pnpm --dir site run data:performance
130131

131-
# Run benchmarks and save results to JSON
132-
bench-save file="bench-results.json":
132+
# Run benchmarks and save results to JSON/Markdown
133+
bench-save file="":
133134
cargo run -p bashkit-bench --release -- --save {{file}}
135+
pnpm --dir site run data:performance
134136

135-
# Run benchmarks with verbose output
137+
# Run benchmarks with verbose output and save site-indexed JSON/Markdown results
136138
bench-verbose:
137-
cargo run -p bashkit-bench --release -- --verbose
139+
cargo run -p bashkit-bench --release -- --verbose --save
140+
pnpm --dir site run data:performance
138141

139-
# Run specific benchmark category (startup, variables, arithmetic, control, strings, arrays, pipes, tools, complex)
142+
# Exploratory: run specific benchmark category without updating site results (startup, variables, arithmetic, control, strings, arrays, pipes, tools, complex)
140143
bench-category cat:
141144
cargo run -p bashkit-bench --release -- --category {{cat}}
142145

143-
# Run benchmarks with more iterations for accuracy
146+
# Run benchmarks with more iterations for accuracy and save site-indexed JSON/Markdown results
144147
bench-accurate:
145-
cargo run -p bashkit-bench --release -- --iterations 50 --warmup 5
148+
cargo run -p bashkit-bench --release -- --iterations 50 --warmup 5 --save
149+
pnpm --dir site run data:performance
146150

147151
# List available benchmarks
148152
bench-list:
149153
cargo run -p bashkit-bench --release -- --list
150154

151-
# Run benchmarks with all runners (including just-bash if available)
155+
# Run benchmarks with all runners and save site-indexed JSON/Markdown results (including just-bash if available)
152156
bench-all:
153-
cargo run -p bashkit-bench --release -- --runners bashkit,bash,just-bash
157+
cargo run -p bashkit-bench --release -- --runners bashkit,bash,just-bash --save
158+
pnpm --dir site run data:performance
154159

155160
# Run Criterion parallel_execution benchmark and save results
156161
bench-parallel:
157162
./scripts/bench-parallel.sh
163+
pnpm --dir site run data:performance
158164

159165
# Run Criterion sqlite builtin benchmark and save results
160166
bench-sqlite:
161167
./scripts/bench-sqlite.sh
168+
pnpm --dir site run data:performance
162169

163170
# === Eval ===
164171

165-
# Run LLM eval (requires ANTHROPIC_API_KEY or OPENAI_API_KEY)
172+
# Run LLM eval and save site-indexed JSON/Markdown results (requires ANTHROPIC_API_KEY or OPENAI_API_KEY)
166173
eval dataset="crates/bashkit-eval/data/eval-tasks.jsonl" provider="anthropic" model="claude-sonnet-4-20250514":
167-
cargo run -p bashkit-eval --release -- run --dataset {{dataset}} --provider {{provider}} --model {{model}}
174+
cargo run -p bashkit-eval --release -- run --dataset {{dataset}} --provider {{provider}} --model {{model}} --save
175+
pnpm --dir site run data:performance
168176

169177
# Run eval and save results
170178
eval-save dataset="crates/bashkit-eval/data/eval-tasks.jsonl" provider="anthropic" model="claude-sonnet-4-20250514":
171179
cargo run -p bashkit-eval --release -- run --dataset {{dataset}} --provider {{provider}} --model {{model}} --save
180+
pnpm --dir site run data:performance
172181

173-
# Run scripting-tool eval (scripted mode)
182+
# Run scripting-tool eval (scripted mode) and save site-indexed JSON/Markdown results
174183
eval-scripting dataset="crates/bashkit-eval/data/scripting-tool/many-tools.jsonl" provider="openai" model="gpt-5.4":
175-
cargo run -p bashkit-eval --release -- run --eval-type scripting-tool --dataset {{dataset}} --provider {{provider}} --model {{model}}
184+
cargo run -p bashkit-eval --release -- run --eval-type scripting-tool --dataset {{dataset}} --provider {{provider}} --model {{model}} --save
185+
pnpm --dir site run data:performance
176186

177-
# Run scripting-tool eval (baseline mode — individual tools, no ScriptedTool)
187+
# Run scripting-tool eval (baseline mode — individual tools, no ScriptedTool) and save site-indexed JSON/Markdown results
178188
eval-scripting-baseline dataset="crates/bashkit-eval/data/scripting-tool/many-tools.jsonl" provider="openai" model="gpt-5.4":
179-
cargo run -p bashkit-eval --release -- run --eval-type scripting-tool --baseline --dataset {{dataset}} --provider {{provider}} --model {{model}}
189+
cargo run -p bashkit-eval --release -- run --eval-type scripting-tool --baseline --dataset {{dataset}} --provider {{provider}} --model {{model}} --save
190+
pnpm --dir site run data:performance
180191

181192
# Run scripting-tool eval and save results
182193
eval-scripting-save dataset="crates/bashkit-eval/data/scripting-tool/many-tools.jsonl" provider="openai" model="gpt-5.4":
183194
cargo run -p bashkit-eval --release -- run --eval-type scripting-tool --dataset {{dataset}} --provider {{provider}} --model {{model}} --save
195+
pnpm --dir site run data:performance
184196

185197
# === Security ===
186198

‎site/README.md‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -18,6 +18,10 @@ pnpm run build # emits ./dist
1818
pnpm run preview # serve dist/ via wrangler
1919
```
2020

21+
`pnpm run build` regenerates `src/data/performance-timeline.json` from saved
22+
benchmark and eval artifacts before Astro builds. The `/benches` page contract is
23+
specified in `../specs/performance-results.md`.
24+
2125
## Deploy
2226

2327
Deployment is intended to run from CI against the Cloudflare account that owns

‎site/package.json‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -10,6 +10,8 @@
1010
},
1111
"scripts": {
1212
"dev": "astro dev",
13+
"data:performance": "node scripts/build-performance-data.mjs",
14+
"prebuild": "node scripts/build-performance-data.mjs",
1315
"build": "astro build",
1416
"postbuild": "node scripts/normalize-generated-html.mjs && node scripts/verify-doc-routes.mjs && node scripts/verify-doc-markdown-routes.mjs && node scripts/verify-public-links.mjs && node scripts/verify-sitemap.mjs && node scripts/verify-robots.mjs && node scripts/verify-agent-skills.mjs && node scripts/verify-link-headers.mjs",
1517
"preview": "wrangler dev",

0 commit comments

Comments
 (0)