From 3c339481d107df82161bff3f4cd18676aab89e84 Mon Sep 17 00:00:00 2001 From: Salil M Date: Wed, 15 Apr 2026 22:34:16 +0530 Subject: [PATCH] Add 'soup bench' command to measure model speed and VRAM usage #24 (#25) * feat(cli): create 'soup bench' command for inference speed and VRAM measurement * register 'bench' command into the main CLI router * add test case for handling missing model paths gracefully * add 'Inference Benchmarking' section explaining the 'soup bench' tool * Added soup.yaml * style: fix linting (unused imports, inconsistent spacing) * style: sort imports in bench and test_bench to satisfy ruff * style: final import sort and grouping fix for CI * Update gitignore --- .gitignore | 2 +- README.md | 14 +++++ soup_cli/cli.py | 2 + soup_cli/commands/bench.py | 126 +++++++++++++++++++++++++++++++++++++ tests/test_bench.py | 14 +++++ 5 files changed, 157 insertions(+), 1 deletion(-) create mode 100644 soup_cli/commands/bench.py create mode 100644 tests/test_bench.py diff --git a/.gitignore b/.gitignore index 31c47fc..52db458 100644 --- a/.gitignore +++ b/.gitignore @@ -52,4 +52,4 @@ report.xml .claude/rules/ .claude/skills/ .claude/settings.json -.coverage +.coverage \ No newline at end of file diff --git a/README.md b/README.md index 262407c..7f6bd75 100644 --- a/README.md +++ b/README.md @@ -861,6 +861,20 @@ soup infer --model ./output --input prompts.jsonl --output results.jsonl \ Output is JSONL with `prompt`, `response`, and `tokens_generated` fields. Shows a progress bar and throughput summary. +## Inference Benchmarking + +Quickly measure your model's generation speed and memory footprint before deployment: + +```bash +# Benchmark local speed and VRAM usage on 3 automatically generated prompts +soup bench ./output + +# Customizing benchmarking parameters +soup bench ./output --num-prompts 5 --max-tokens 256 +``` + +This acts as a built-in "speedometer," outputting Tokens-Per-Second (TPS), Total Latency, and Peak VRAM allocations into a clean status table. + ## TensorBoard Integration Log training metrics to TensorBoard for local visualization: diff --git a/soup_cli/cli.py b/soup_cli/cli.py index 76c1fe5..3fc4bd6 100644 --- a/soup_cli/cli.py +++ b/soup_cli/cli.py @@ -9,6 +9,7 @@ from soup_cli import __version__ from soup_cli.commands import ( adapters, autopilot, + bench, chat, data, deploy, @@ -82,6 +83,7 @@ app.command()(sweep.sweep) app.command(name="diff")(diff.diff) app.command()(infer.infer) app.command()(profile.profile) +app.command()(bench.bench) app.command()(doctor_cmd.doctor) app.command()(quickstart_cmd.quickstart) app.command()(ui.ui) diff --git a/soup_cli/commands/bench.py b/soup_cli/commands/bench.py new file mode 100644 index 0000000..6da0d9a --- /dev/null +++ b/soup_cli/commands/bench.py @@ -0,0 +1,126 @@ +"""soup bench — simple measuring tool for model speed and memory.""" + +import time +from pathlib import Path +from typing import Optional + +import typer +from rich.console import Console +from rich.panel import Panel +from rich.table import Table + +console = Console() + + +def bench( + model: str = typer.Argument( + ..., + help="Path to model (LoRA adapter or full model) to benchmark", + ), + base: Optional[str] = typer.Option( + None, + "--base", + "-b", + help="Base model for LoRA adapter (auto-detected if not set)", + ), + max_tokens: int = typer.Option( + 128, + "--max-tokens", + help="Maximum tokens to generate per prompt", + ), + num_prompts: int = typer.Option( + 3, + "--num-prompts", + "-n", + help="Number of prompts to run for averaging", + ), +): + """Run an inference benchmark (speed and memory) on a loaded model.""" + import torch + + from soup_cli.commands.infer import _generate, _load_model + from soup_cli.utils.gpu import detect_device + + model_path = Path(model) + if not model_path.exists(): + console.print(f"[red]Model not found: {model_path}[/]") + raise typer.Exit(1) + + device, _ = detect_device() + + console.print( + Panel( + f"Model: [bold]{model_path}[/]\n" + f"Device: [bold]{device}[/]\n" + f"Prompts: [bold]{num_prompts}[/]\n" + f"Tokens/P: [bold]{max_tokens}[/]", + title="Benchmarking Configuration", + ) + ) + + console.print("[dim]Loading model to measure resource usage...[/]") + + if torch.cuda.is_available(): + torch.cuda.reset_peak_memory_stats() + + start_load = time.time() + try: + model_obj, tokenizer = _load_model(str(model_path), base, device) + except Exception as exc: + console.print(f"[red]Failed to load model:[/] {exc}") + raise typer.Exit(1) + + load_time = time.time() - start_load + console.print(f"[green]Model loaded in {load_time:.2f}s.[/]\n") + + prompts = [ + "Explain the theory of relativity briefly.", + "Write a short Python function to calculate fibonacci numbers.", + "What are the main consequences of the Industrial Revolution?", + "Compose a poem about a wandering space traveler.", + "Describe how a database index works under the hood." + ] + # Loop over prompts + test_prompts = (prompts * (num_prompts // len(prompts) + 1))[:num_prompts] + + total_tokens = 0 + total_latency = 0.0 + + console.print(f"[bold]Running {num_prompts} test inferences...[/]") + + for i, prompt_text in enumerate(test_prompts): + messages = [{"role": "user", "content": prompt_text}] + start_time = time.time() + + response, token_count = _generate( + model_obj, tokenizer, messages, + max_tokens=max_tokens, temperature=0.0, + ) + + latency = time.time() - start_time + total_tokens += token_count + total_latency += latency + console.print(f" [dim]Prompt {i+1}: {token_count} tokens in {latency:.2f}s[/]") + + avg_tps = total_tokens / total_latency if total_latency > 0 else 0 + + peak_vram_gb = 0.0 + if torch.cuda.is_available(): + peak_vram_gb = torch.cuda.max_memory_allocated() / (1024**3) + + table = Table(title="Inference Benchmark Results") + table.add_column("Backend", style="cyan") + table.add_column("TPS (Avg)", style="green", justify="right") + table.add_column("Latency (Total)", style="yellow", justify="right") + table.add_column("Max VRAM", style="magenta", justify="right") + + vram_str = f"{peak_vram_gb:.2f} GB" if torch.cuda.is_available() else "N/A" + table.add_row( + "Transformers", + f"{avg_tps:.2f}", + f"{total_latency:.2f}s", + vram_str + ) + + console.print() + console.print(table) diff --git a/tests/test_bench.py b/tests/test_bench.py new file mode 100644 index 0000000..6af3e3e --- /dev/null +++ b/tests/test_bench.py @@ -0,0 +1,14 @@ +"""Tests for soup bench CLI command.""" + +from typer.testing import CliRunner + +from soup_cli.cli import app + +runner = CliRunner() + + +def test_bench_model_not_found(): + """soup bench with nonexistent model should fail gracefully.""" + result = runner.invoke(app, ["bench", "nonexistent_model_path"]) + assert result.exit_code == 1 + assert "not found" in result.output.lower()