# modal_bench.py — deploy: pip install modal && modal deploy modal_bench.py
# Starter credits can cover short runs; review current pricing and set a spend budget.
import time
import modal
app = modal.App("mp-gpu-bench")
image = modal.Image.debian_slim().pip_install("torch", "numpy", "fastapi[standard]")
@app.function(image=image, gpu="A10G", timeout=120)
@modal.fastapi_endpoint(method="POST")
def bench(req: dict):
if req.get("ping"):
return {"ok": True, "t": time.time()} # same GPU container pool as the bench
import torch
size = int(req.get("size", 2048))
iters = int(req.get("iters", 20))
dev = torch.device("cuda")
a = torch.randn(size, size, device=dev)
b = torch.randn(size, size, device=dev)
torch.cuda.synchronize()
for _ in range(3):
(a @ b)
torch.cuda.synchronize()
t0 = time.perf_counter()
for _ in range(iters):
c = a @ b
torch.cuda.synchronize()
dt = time.perf_counter() - t0
flops = 2 * size**3 * iters
return {
"gpu": torch.cuda.get_device_name(0),
"vram_gb": round(torch.cuda.get_device_properties(0).total_memory / 1e9, 1),
"size": size, "iters": iters,
"seconds": round(dt, 4),
"tflops": round(flops / dt / 1e12, 2),
"checksum": float(c[0, 0]), # proves the multiply actually ran
}
# Optional: real Whisper-large-v3 transcription endpoint (adds ~3 GB image):
# image2 = image.pip_install("openai-whisper")
# @app.function(image=image2, gpu="A10G", timeout=300)
# @modal.fastapi_endpoint(method="POST")
# def transcribe(req: dict):
# import base64, tempfile, whisper
# model = whisper.load_model("large-v3")
# with tempfile.NamedTemporaryFile(suffix=".wav") as f:
# f.write(base64.b64decode(req["wav_b64"])); f.flush()
# t0 = time.perf_counter()
# out = model.transcribe(f.name)
# return {"text": out["text"], "seconds": time.perf_counter() - t0}