# Benchmark VRAM/czas: BiRefNet (sam matte) + pelna sciezka segment() z negatywami. # Uruchom: python tmp/bench_birefnet.py [A|B|C] (cwd = vilmal/) # Fazy jako osobne procesy (host RAM ~8 GB) — wynik JSON na stdout per faza. import gc import json import sys import time import numpy as np import torch from PIL import Image sys.path.insert(0, r"C:\xampp\htdocs\vilmax\vilmal\services\recolor") IMG_PREVIEW = r"C:\xampp\htdocs\vilmax\vilmax-cockpit\data\hall-audit\preview\p-6419.jpg" # 1024x768 IMG_FULL = r"C:\xampp\htdocs\vilmax\vilmax-cockpit\data\hall-audit\img-6419.jpg" # 5712x4284 def vram(): return { "alloc_mb": round(torch.cuda.memory_allocated() / 2**20, 1), "reserved_mb": round(torch.cuda.memory_reserved() / 2**20, 1), "peak_mb": round(torch.cuda.max_memory_allocated() / 2**20, 1), } def phase_a(): """Sam BiRefNet: load + forward 1024x1024 (stale wejscie modelu w _matte).""" from transformers import AutoModelForImageSegmentation torch.cuda.reset_peak_memory_stats() t0 = time.time() model = AutoModelForImageSegmentation.from_pretrained( "ZhengPeng7/BiRefNet", trust_remote_code=True ).to("cuda") model.eval() load_s = time.time() - t0 after_load = vram() n_params = sum(p.numel() for p in model.parameters()) x = torch.rand(1, 3, 1024, 1024, device="cuda") with torch.no_grad(): model(x) torch.cuda.synchronize() torch.cuda.reset_peak_memory_stats() times = [] with torch.no_grad(): for _ in range(5): t = time.time() model(x) torch.cuda.synchronize() times.append(round((time.time() - t) * 1000, 1)) return { "params_m": round(n_params / 1e6, 1), "load_s": round(load_s, 2), "vram_after_load": after_load, "forward_1024x1024_ms": times, "vram_forward_peak": vram(), } def phase_bc(size): """Pelna segment() (BiRefNet + DINO/SAM negatywy) na realnym zdjeciu.""" from fabric_recolor.segmentation.birefnet import BirefnetMatteSegmenter from fabric_recolor.segmentation.prompts import NEGATIVE_PROMPTS, POSITIVE_PROMPTS src = Image.open(IMG_FULL).convert("RGB").resize(size, Image.BILINEAR) rgb = np.asarray(src, dtype=np.uint8) del src torch.cuda.reset_peak_memory_stats() seg = BirefnetMatteSegmenter(device="cuda", max_side=1600) t0 = time.time() res = seg.segment(rgb, POSITIVE_PROMPTS, NEGATIVE_PROMPTS) seg_s = time.time() - t0 cov = float((res.alpha > 10 / 255).mean()) r = { "image": list(size), "elapsed_s": round(seg_s, 2), "vram_peak": vram(), "coverage_gt10": round(cov, 4), "notes": res.notes, } del rgb, res, seg return r if __name__ == "__main__": assert torch.cuda.is_available() which = sys.argv[1] if len(sys.argv) > 1 else "A" out = {"gpu": torch.cuda.get_device_name(0), "torch": torch.__version__} if which == "A": out["birefnet"] = phase_a() elif which == "B": out["full_segment_1600x1200"] = phase_bc((1600, 1200)) elif which == "C": out["full_segment_1024x768"] = phase_bc((1024, 768)) print(json.dumps(out, indent=2, ensure_ascii=False))