# Benchmark VRAM/czas: Depth-Anything-V2-Small-hf przez transformers (fp32 + fp16). # Uruchom: python tmp/bench_depth.py (cwd = vilmal/) import gc import json import time import torch from PIL import Image IMG_PREVIEW = r"C:\xampp\htdocs\vilmax\vilmax-cockpit\data\hall-audit\preview\p-6419.jpg" # 1024x768 MODEL_ID = "depth-anything/Depth-Anything-V2-Small-hf" def vram(): return { "alloc_mb": round(torch.cuda.memory_allocated() / 2**20, 1), "reserved_mb": round(torch.cuda.memory_reserved() / 2**20, 1), "peak_mb": round(torch.cuda.max_memory_allocated() / 2**20, 1), } def run(model, proc, image, dev, n=5): inputs = proc(images=image, return_tensors="pt").to(dev) if next(model.parameters()).dtype == torch.float16: inputs = {k: (v.half() if v.dtype == torch.float32 else v) for k, v in inputs.items()} in_h, in_w = int(inputs["pixel_values"].shape[2]), int(inputs["pixel_values"].shape[3]) with torch.no_grad(): model(**inputs) # warmup torch.cuda.synchronize() torch.cuda.reset_peak_memory_stats() times = [] with torch.no_grad(): for _ in range(n): t = time.time() out = model(**inputs) torch.cuda.synchronize() times.append(round((time.time() - t) * 1000, 1)) depth = out.predicted_depth return { "input_hw": [in_h, in_w], "out_hw": list(depth.shape[-2:]), "forward_ms": times, "vram_peak": vram(), } def main(): assert torch.cuda.is_available() from transformers import AutoImageProcessor, AutoModelForDepthEstimation dev = "cuda" image = Image.open(IMG_PREVIEW).convert("RGB") out = {"gpu": torch.cuda.get_device_name(0), "model": MODEL_ID} t0 = time.time() proc = AutoImageProcessor.from_pretrained(MODEL_ID) print("proc_ok", flush=True) # ---------- fp32 ---------- torch.cuda.empty_cache() torch.cuda.reset_peak_memory_stats() t0 = time.time() model = AutoModelForDepthEstimation.from_pretrained(MODEL_ID).to(dev) model.eval() load_s = time.time() - t0 n_params = sum(p.numel() for p in model.parameters()) after_load = vram() r = run(model, proc, image, dev) out["fp32"] = { "params_m": round(n_params / 1e6, 1), "load_s": round(load_s, 2), "vram_after_load": after_load, **r, } del model gc.collect() torch.cuda.empty_cache() # ---------- fp16 ---------- torch.cuda.reset_peak_memory_stats() t0 = time.time() model = AutoModelForDepthEstimation.from_pretrained(MODEL_ID, torch_dtype=torch.float16).to(dev) model.eval() load_s = time.time() - t0 after_load = vram() r = run(model, proc, image, dev) out["fp16"] = { "load_s": round(load_s, 2), "vram_after_load": after_load, **r, } del model gc.collect() torch.cuda.empty_cache() print(json.dumps(out, indent=2, ensure_ascii=False)) if __name__ == "__main__": main()