mirror of https://github.com/razor-ai/soup.git
* feat(cli): create 'soup bench' command for inference speed and VRAM measurement * register 'bench' command into the main CLI router * add test case for handling missing model paths gracefully * add 'Inference Benchmarking' section explaining the 'soup bench' tool * Added soup.yaml * style: fix linting (unused imports, inconsistent spacing) * style: sort imports in bench and test_bench to satisfy ruff * style: final import sort and grouping fix for CI * Update gitignore
This commit is contained in:
parent
14e86da0e4
commit
3c339481d1
|
|
@ -52,4 +52,4 @@ report.xml
|
|||
.claude/rules/
|
||||
.claude/skills/
|
||||
.claude/settings.json
|
||||
.coverage
|
||||
.coverage
|
||||
14
README.md
14
README.md
|
|
@ -861,6 +861,20 @@ soup infer --model ./output --input prompts.jsonl --output results.jsonl \
|
|||
|
||||
Output is JSONL with `prompt`, `response`, and `tokens_generated` fields. Shows a progress bar and throughput summary.
|
||||
|
||||
## Inference Benchmarking
|
||||
|
||||
Quickly measure your model's generation speed and memory footprint before deployment:
|
||||
|
||||
```bash
|
||||
# Benchmark local speed and VRAM usage on 3 automatically generated prompts
|
||||
soup bench ./output
|
||||
|
||||
# Customizing benchmarking parameters
|
||||
soup bench ./output --num-prompts 5 --max-tokens 256
|
||||
```
|
||||
|
||||
This acts as a built-in "speedometer," outputting Tokens-Per-Second (TPS), Total Latency, and Peak VRAM allocations into a clean status table.
|
||||
|
||||
## TensorBoard Integration
|
||||
|
||||
Log training metrics to TensorBoard for local visualization:
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ from soup_cli import __version__
|
|||
from soup_cli.commands import (
|
||||
adapters,
|
||||
autopilot,
|
||||
bench,
|
||||
chat,
|
||||
data,
|
||||
deploy,
|
||||
|
|
@ -82,6 +83,7 @@ app.command()(sweep.sweep)
|
|||
app.command(name="diff")(diff.diff)
|
||||
app.command()(infer.infer)
|
||||
app.command()(profile.profile)
|
||||
app.command()(bench.bench)
|
||||
app.command()(doctor_cmd.doctor)
|
||||
app.command()(quickstart_cmd.quickstart)
|
||||
app.command()(ui.ui)
|
||||
|
|
|
|||
|
|
@ -0,0 +1,126 @@
|
|||
"""soup bench — simple measuring tool for model speed and memory."""
|
||||
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
from rich.console import Console
|
||||
from rich.panel import Panel
|
||||
from rich.table import Table
|
||||
|
||||
console = Console()
|
||||
|
||||
|
||||
def bench(
|
||||
model: str = typer.Argument(
|
||||
...,
|
||||
help="Path to model (LoRA adapter or full model) to benchmark",
|
||||
),
|
||||
base: Optional[str] = typer.Option(
|
||||
None,
|
||||
"--base",
|
||||
"-b",
|
||||
help="Base model for LoRA adapter (auto-detected if not set)",
|
||||
),
|
||||
max_tokens: int = typer.Option(
|
||||
128,
|
||||
"--max-tokens",
|
||||
help="Maximum tokens to generate per prompt",
|
||||
),
|
||||
num_prompts: int = typer.Option(
|
||||
3,
|
||||
"--num-prompts",
|
||||
"-n",
|
||||
help="Number of prompts to run for averaging",
|
||||
),
|
||||
):
|
||||
"""Run an inference benchmark (speed and memory) on a loaded model."""
|
||||
import torch
|
||||
|
||||
from soup_cli.commands.infer import _generate, _load_model
|
||||
from soup_cli.utils.gpu import detect_device
|
||||
|
||||
model_path = Path(model)
|
||||
if not model_path.exists():
|
||||
console.print(f"[red]Model not found: {model_path}[/]")
|
||||
raise typer.Exit(1)
|
||||
|
||||
device, _ = detect_device()
|
||||
|
||||
console.print(
|
||||
Panel(
|
||||
f"Model: [bold]{model_path}[/]\n"
|
||||
f"Device: [bold]{device}[/]\n"
|
||||
f"Prompts: [bold]{num_prompts}[/]\n"
|
||||
f"Tokens/P: [bold]{max_tokens}[/]",
|
||||
title="Benchmarking Configuration",
|
||||
)
|
||||
)
|
||||
|
||||
console.print("[dim]Loading model to measure resource usage...[/]")
|
||||
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.reset_peak_memory_stats()
|
||||
|
||||
start_load = time.time()
|
||||
try:
|
||||
model_obj, tokenizer = _load_model(str(model_path), base, device)
|
||||
except Exception as exc:
|
||||
console.print(f"[red]Failed to load model:[/] {exc}")
|
||||
raise typer.Exit(1)
|
||||
|
||||
load_time = time.time() - start_load
|
||||
console.print(f"[green]Model loaded in {load_time:.2f}s.[/]\n")
|
||||
|
||||
prompts = [
|
||||
"Explain the theory of relativity briefly.",
|
||||
"Write a short Python function to calculate fibonacci numbers.",
|
||||
"What are the main consequences of the Industrial Revolution?",
|
||||
"Compose a poem about a wandering space traveler.",
|
||||
"Describe how a database index works under the hood."
|
||||
]
|
||||
# Loop over prompts
|
||||
test_prompts = (prompts * (num_prompts // len(prompts) + 1))[:num_prompts]
|
||||
|
||||
total_tokens = 0
|
||||
total_latency = 0.0
|
||||
|
||||
console.print(f"[bold]Running {num_prompts} test inferences...[/]")
|
||||
|
||||
for i, prompt_text in enumerate(test_prompts):
|
||||
messages = [{"role": "user", "content": prompt_text}]
|
||||
start_time = time.time()
|
||||
|
||||
response, token_count = _generate(
|
||||
model_obj, tokenizer, messages,
|
||||
max_tokens=max_tokens, temperature=0.0,
|
||||
)
|
||||
|
||||
latency = time.time() - start_time
|
||||
total_tokens += token_count
|
||||
total_latency += latency
|
||||
console.print(f" [dim]Prompt {i+1}: {token_count} tokens in {latency:.2f}s[/]")
|
||||
|
||||
avg_tps = total_tokens / total_latency if total_latency > 0 else 0
|
||||
|
||||
peak_vram_gb = 0.0
|
||||
if torch.cuda.is_available():
|
||||
peak_vram_gb = torch.cuda.max_memory_allocated() / (1024**3)
|
||||
|
||||
table = Table(title="Inference Benchmark Results")
|
||||
table.add_column("Backend", style="cyan")
|
||||
table.add_column("TPS (Avg)", style="green", justify="right")
|
||||
table.add_column("Latency (Total)", style="yellow", justify="right")
|
||||
table.add_column("Max VRAM", style="magenta", justify="right")
|
||||
|
||||
vram_str = f"{peak_vram_gb:.2f} GB" if torch.cuda.is_available() else "N/A"
|
||||
table.add_row(
|
||||
"Transformers",
|
||||
f"{avg_tps:.2f}",
|
||||
f"{total_latency:.2f}s",
|
||||
vram_str
|
||||
)
|
||||
|
||||
console.print()
|
||||
console.print(table)
|
||||
|
|
@ -0,0 +1,14 @@
|
|||
"""Tests for soup bench CLI command."""
|
||||
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from soup_cli.cli import app
|
||||
|
||||
runner = CliRunner()
|
||||
|
||||
|
||||
def test_bench_model_not_found():
|
||||
"""soup bench with nonexistent model should fail gracefully."""
|
||||
result = runner.invoke(app, ["bench", "nonexistent_model_path"])
|
||||
assert result.exit_code == 1
|
||||
assert "not found" in result.output.lower()
|
||||
Loading…
Reference in New Issue