mirror of
https://github.com/hacksider/Deep-Live-Cam.git
synced 2026-09-05 07:16:35 +02:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bc48ba4308 | ||
|
|
b53844eb57 | ||
|
|
20cafb0079 | ||
|
|
7ca6d0b202 | ||
|
|
f7db37679a | ||
|
|
bd500d54b4 | ||
|
|
987f6b392b | ||
|
|
97a44800a2 | ||
|
|
345fa4a0b0 | ||
|
|
bdeeb3ace0 | ||
|
|
230217ec11 | ||
|
|
156321f7a3 | ||
|
|
8234965ee8 | ||
|
|
ab64c186ec | ||
|
|
b8e781e539 | ||
|
|
ff7ee0d219 | ||
|
|
8d727eba3e | ||
|
|
eba2a958d3 | ||
|
|
14ba4f9c0b | ||
|
|
7d2d7fb1f3 | ||
|
|
57c4c32377 | ||
|
|
d00b09f5d8 | ||
|
|
897dc21da4 | ||
|
|
834092c891 | ||
|
|
da0672ad6b | ||
|
|
47dffeb307 | ||
|
|
9e1f0cc3a5 | ||
|
|
834bc43768 | ||
|
|
3b69413d61 | ||
|
|
07e2e960c8 | ||
|
|
ba27b75265 | ||
|
|
cfa8123b67 | ||
|
|
08b2dd2526 | ||
|
|
886e64b320 | ||
|
|
aa6f2cbade | ||
|
|
0b61ad5c0d | ||
|
|
a21ccf488c | ||
|
|
ca8e39e3bb | ||
|
|
0e97e474e4 | ||
|
|
9c67a7aacc | ||
|
|
4a674d33ef | ||
|
|
682450755f | ||
|
|
12a3f6a007 | ||
|
|
cede099ccb | ||
|
|
81a1986ef8 | ||
|
|
ed758eb693 | ||
|
|
9c5f01c7f1 | ||
|
|
8bdc348779 | ||
|
|
e34d204c2e | ||
|
|
d1376b07d1 | ||
|
|
5deadaf428 | ||
|
|
2fba52e11b | ||
|
|
0926b65aaf | ||
|
|
297acded3b | ||
|
|
014bce0704 | ||
|
|
c962399669 | ||
|
|
2dd42dfc75 | ||
|
|
c38d669f7c | ||
|
|
890a6d41b6 | ||
|
|
f95a0bb7fb | ||
|
|
e957a7f4dd | ||
|
|
19416cb3cb | ||
|
|
cbf0859347 | ||
|
|
a6c99607fc | ||
|
|
0a87d63560 | ||
|
|
ea19030c74 | ||
|
|
4d04e830bc | ||
|
|
f65aeae5db | ||
|
|
64d3f06089 | ||
|
|
fceafcb234 | ||
|
|
033475b89c | ||
|
|
07711af712 | ||
|
|
44664d8a7f | ||
|
|
15a3f537a4 | ||
|
|
fbcea9e135 | ||
|
|
646b0f816f | ||
|
|
bcdd0ce2dd | ||
|
|
8703d394d6 | ||
|
|
69e3fc5611 | ||
|
|
2b26d5539e | ||
|
|
fea5a4c2d2 | ||
|
|
51fb7a6ad6 | ||
|
|
6da4f398d5 | ||
|
|
3e362383d8 | ||
|
|
11fb5bfbc6 | ||
|
|
586d8f3fb0 | ||
|
|
1edc4bc298 | ||
|
|
1f3668f7c1 | ||
|
|
3d16ee346f | ||
|
|
ab834d5640 | ||
|
|
bf8a89d20a | ||
|
|
bb4ef4a133 | ||
|
|
b6b6c741a2 | ||
|
|
1b240a45fd | ||
|
|
ecf02d0640 | ||
|
|
0cbc9f126f | ||
|
|
a3fd56a312 | ||
|
|
9525d45291 | ||
|
|
d9a5500bdf | ||
|
|
86134b6e1d | ||
|
|
fbd1cc5973 | ||
|
|
eac2ad2307 | ||
|
|
9e6f30c0a4 | ||
|
|
97321a740d | ||
|
|
9207386e07 | ||
|
|
f5f7ac7764 | ||
|
|
77d3492eef | ||
|
|
8e3d6e7c65 | ||
|
|
ee9699ee70 | ||
|
|
3c8b259a3f | ||
|
|
30b27c2b71 |
@@ -0,0 +1,16 @@
|
|||||||
|
name: ruff
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request:
|
||||||
|
push:
|
||||||
|
branches: [main]
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
ruff:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- uses: astral-sh/ruff-action@v4.0.0
|
||||||
|
with:
|
||||||
|
version: "0.15.7"
|
||||||
|
args: "check --output-format=github"
|
||||||
@@ -27,3 +27,5 @@ faceswap/
|
|||||||
switch_states.json
|
switch_states.json
|
||||||
/models
|
/models
|
||||||
install.bat
|
install.bat
|
||||||
|
/.claude
|
||||||
|
*.bat
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
<h1 align="center">Deep-Live-Cam 2.0.5c</h1>
|
<h1 align="center">Deep-Live-Cam 2.1.6</h1>
|
||||||
|
|
||||||
<p align="center">
|
<p align="center">
|
||||||
Real-time face swap and video deepfake with a single click and only a single image.
|
Real-time face swap and video deepfake with a single click and only a single image.
|
||||||
@@ -30,13 +30,45 @@ By using this software, you agree to these terms and commit to using it in a man
|
|||||||
|
|
||||||
Users are expected to use this software responsibly and legally. If using a real person's face, obtain their consent and clearly label any output as a deepfake when sharing online. We are not responsible for end-user actions.
|
Users are expected to use this software responsibly and legally. If using a real person's face, obtain their consent and clearly label any output as a deepfake when sharing online. We are not responsible for end-user actions.
|
||||||
|
|
||||||
## Exclusive v2.6d Quick Start - Pre-built (Windows/Mac Silicon)
|
## Pre-built Deep-Live-Cam 2.7 Ultimate!
|
||||||
|
|
||||||
<a href="https://deeplivecam.net/index.php/quickstart"> <img src="media/Download.png" width="285" height="77" />
|
<p align="center">
|
||||||
|
<a href="https://deeplivecam.net/index.php/quickstart">
|
||||||
|
<img src="https://github.com/user-attachments/assets/fa2cdf79-c933-4b93-844a-b087192261ed" width="100%" alt="Lite / Ultimate Download Banner">
|
||||||
|
</a>
|
||||||
|
</p>
|
||||||
|
|
||||||
##### This is the fastest build you can get if you have a discrete NVIDIA or AMD GPU or Mac Silicon, And you'll receive special priority support.
|
<p align="center">
|
||||||
|
<a href="https://deeplivecam.net/index.php/plans/nvidia-gpu?plan_id=0&group_id=1">
|
||||||
###### These Pre-builts are perfect for non-technical users or those who don't have time to, or can't manually install all the requirements. Just a heads-up: this is an open-source project, so you can also install it manually.
|
<img src="https://github.com/user-attachments/assets/56b61811-3a1e-4672-9b50-cf7f6e8e6852" width="40" alt="Windows">
|
||||||
|
</a>
|
||||||
|
|
||||||
|
<a href="https://deeplivecam.net/index.php/plans/nvidia-gpu?plan_id=0&group_id=2">
|
||||||
|
<img src="https://github.com/user-attachments/assets/6538e3a6-c957-431a-b586-2d6abcf534dc" width="34" alt="Mac Silicon">
|
||||||
|
</a>
|
||||||
|
|
||||||
|
<a href="https://deeplivecam.net/index.php/plans/nvidia-gpu?plan_id=0&group_id=3">
|
||||||
|
<img src="https://github.com/user-attachments/assets/ad45142e-426c-4364-a2a9-a512670cc62c" width="40" alt="CPU">
|
||||||
|
</a>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p align="center">
|
||||||
|
<strong>Windows • Mac Silicon • CPU • NVIDIA • AMD</strong>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p align="center">
|
||||||
|
Builds optimized for your hardware.
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p align="center">
|
||||||
|
<a href="https://deeplivecam.net/index.php/quickstart">
|
||||||
|
<img src="media/Download.png" width="280" alt="Download">
|
||||||
|
</a>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
> **Ultimate** includes **30+ exclusive features**, performance optimizations, and **priority support** We only have a single official website which is https://deeplivecam.net . Please be careful on where you download other versions of this application aside from that website and this github repo.
|
||||||
|
|
||||||
|
Perfect if you want the fastest setup with **zero manual installation**, pre-configured dependencies, and optimized builds for every supported platform.
|
||||||
|
|
||||||
## TLDR; Live Deepfake in just 3 Clicks
|
## TLDR; Live Deepfake in just 3 Clicks
|
||||||

|

|
||||||
@@ -109,7 +141,7 @@ This is more likely to work on your computer but will be slower as it utilizes t
|
|||||||
|
|
||||||
**1. Set up Your Platform**
|
**1. Set up Your Platform**
|
||||||
|
|
||||||
- Python (3.11 recommended)
|
- Python (3.14 recommended; 3.11-3.14 supported)
|
||||||
- pip
|
- pip
|
||||||
- git
|
- git
|
||||||
- [ffmpeg](https://www.youtube.com/watch?v=OlNWCpFdVMA) - ```iex (irm ffmpeg.tc.ht)```
|
- [ffmpeg](https://www.youtube.com/watch?v=OlNWCpFdVMA) - ```iex (irm ffmpeg.tc.ht)```
|
||||||
@@ -118,13 +150,13 @@ This is more likely to work on your computer but will be slower as it utilizes t
|
|||||||
**2. Clone the Repository**
|
**2. Clone the Repository**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone https://github.com/hacksider/Deep-Live-Cam.git
|
git clone --depth 1 https://github.com/hacksider/Deep-Live-Cam.git
|
||||||
cd Deep-Live-Cam
|
cd Deep-Live-Cam
|
||||||
```
|
```
|
||||||
|
|
||||||
**3. Download the Models**
|
**3. Download the Models**
|
||||||
|
|
||||||
1. [GFPGANv1.4](https://huggingface.co/hacksider/deep-live-cam/resolve/main/GFPGANv1.4.onnx)
|
1. [gfpgan-1024.onnx](https://huggingface.co/hacksider/deep-live-cam/resolve/main/gfpgan-1024.onnx)
|
||||||
2. [inswapper\_128\_fp16.onnx](https://huggingface.co/hacksider/deep-live-cam/resolve/main/inswapper_128_fp16.onnx)
|
2. [inswapper\_128\_fp16.onnx](https://huggingface.co/hacksider/deep-live-cam/resolve/main/inswapper_128_fp16.onnx)
|
||||||
|
|
||||||
Place these files in the "**models**" folder.
|
Place these files in the "**models**" folder.
|
||||||
@@ -142,7 +174,7 @@ pip install -r requirements.txt
|
|||||||
```
|
```
|
||||||
For Linux:
|
For Linux:
|
||||||
```bash
|
```bash
|
||||||
# Ensure you use the installed Python 3.10
|
# Ensure you use the installed Python 3.14
|
||||||
python3 -m venv venv
|
python3 -m venv venv
|
||||||
source venv/bin/activate
|
source venv/bin/activate
|
||||||
pip install -r requirements.txt
|
pip install -r requirements.txt
|
||||||
@@ -150,17 +182,17 @@ pip install -r requirements.txt
|
|||||||
|
|
||||||
**For macOS:**
|
**For macOS:**
|
||||||
|
|
||||||
Apple Silicon (M1/M2/M3) requires specific setup:
|
Apple Silicon (M1 through M5) requires specific setup:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Install Python 3.11 (specific version is important)
|
# Install Python 3.14
|
||||||
brew install python@3.11
|
brew install python@3.14
|
||||||
|
|
||||||
# Install tkinter package (required for the GUI)
|
# Install tkinter package (required for the GUI)
|
||||||
brew install python-tk@3.10
|
brew install python-tk@3.14
|
||||||
|
|
||||||
# Create and activate virtual environment with Python 3.11
|
# Create and activate virtual environment with Python 3.14
|
||||||
python3.11 -m venv venv
|
python3.14 -m venv venv
|
||||||
source venv/bin/activate
|
source venv/bin/activate
|
||||||
|
|
||||||
# Install dependencies
|
# Install dependencies
|
||||||
@@ -201,7 +233,7 @@ pip install git+https://github.com/TencentARC/GFPGAN.git@master
|
|||||||
```bash
|
```bash
|
||||||
pip install -U torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu128
|
pip install -U torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu128
|
||||||
pip uninstall onnxruntime onnxruntime-gpu
|
pip uninstall onnxruntime onnxruntime-gpu
|
||||||
pip install onnxruntime-gpu==1.21.0
|
pip install onnxruntime-gpu==1.26.0
|
||||||
```
|
```
|
||||||
|
|
||||||
3. Usage:
|
3. Usage:
|
||||||
@@ -212,36 +244,39 @@ python run.py --execution-provider cuda
|
|||||||
|
|
||||||
**CoreML Execution Provider (Apple Silicon)**
|
**CoreML Execution Provider (Apple Silicon)**
|
||||||
|
|
||||||
Apple Silicon (M1/M2/M3) specific installation:
|
Apple Silicon (M1 through M5) specific installation:
|
||||||
|
|
||||||
1. Make sure you've completed the macOS setup above using Python 3.10.
|
1. Make sure you've completed the macOS setup above using Python 3.14.
|
||||||
2. Install dependencies:
|
2. No extra install step is needed — `requirements.txt` pulls the official
|
||||||
|
`onnxruntime` build, whose macOS wheels ship the CoreML execution provider.
|
||||||
|
If you previously installed the unmaintained `onnxruntime-silicon` fork,
|
||||||
|
remove it first, as it shadows the real package:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pip uninstall onnxruntime onnxruntime-silicon
|
pip uninstall onnxruntime-silicon
|
||||||
pip install onnxruntime-silicon==1.13.1
|
pip install -r requirements.txt
|
||||||
```
|
```
|
||||||
|
|
||||||
3. Usage (important: specify Python 3.10):
|
3. Usage:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
python3.10 run.py --execution-provider coreml
|
python3.14 run.py --execution-provider coreml
|
||||||
```
|
```
|
||||||
|
|
||||||
**Important Notes for macOS:**
|
**Important Notes for macOS:**
|
||||||
- You **must** use Python 3.10, not newer versions like 3.11 or 3.13
|
- Python 3.11 is the minimum (onnxruntime dropped 3.10); 3.14 is recommended
|
||||||
- Always run with `python3.10` command not just `python` if you have multiple Python versions installed
|
- Always run with `python3.14` command not just `python` if you have multiple Python versions installed
|
||||||
- If you get error about `_tkinter` missing, reinstall the tkinter package: `brew reinstall python-tk@3.10`
|
- If you get error about `_tkinter` missing, reinstall the tkinter package: `brew reinstall python-tk@3.14`
|
||||||
- If you get model loading errors, check that your models are in the correct folder
|
- If you get model loading errors, check that your models are in the correct folder
|
||||||
- If you encounter conflicts with other Python versions, consider uninstalling them:
|
- If you encounter conflicts with other Python versions, consider uninstalling them:
|
||||||
```bash
|
```bash
|
||||||
# List all installed Python versions
|
# List all installed Python versions
|
||||||
brew list | grep python
|
brew list | grep python
|
||||||
|
|
||||||
# Uninstall conflicting versions if needed
|
# Uninstall conflicting versions if needed
|
||||||
brew uninstall --ignore-dependencies python@3.11 python@3.13
|
brew uninstall --ignore-dependencies python@3.11
|
||||||
|
|
||||||
# Keep only Python 3.11
|
# Keep only Python 3.14
|
||||||
brew cleanup
|
brew cleanup
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -284,6 +319,22 @@ pip uninstall onnxruntime onnxruntime-openvino
|
|||||||
pip install onnxruntime-openvino==1.21.0
|
pip install onnxruntime-openvino==1.21.0
|
||||||
```
|
```
|
||||||
|
|
||||||
|
**Note:** `onnxruntime-openvino` newer than 1.21.0 must be installed together with `openvino`, and the two versions must correspond one-to-one. The supported pairings are:
|
||||||
|
|
||||||
|
| onnxruntime-openvino | OpenVINO |
|
||||||
|
| --- | --- |
|
||||||
|
| 1.24.1 | 2025.4.1 |
|
||||||
|
| 1.23.0 | 2025.3 |
|
||||||
|
| 1.22.0 | 2025.1 |
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Example: onnxruntime-openvino 1.24.1 pairs with OpenVINO 2025.4.1
|
||||||
|
pip install openvino==2025.4.1
|
||||||
|
pip install onnxruntime-openvino==1.24.1
|
||||||
|
```
|
||||||
|
|
||||||
|
See the [OpenVINO Execution Provider requirements](https://onnxruntime.ai/docs/execution-providers/OpenVINO-ExecutionProvider.html#requirements) for the full version-mapping details.
|
||||||
|
|
||||||
2. Usage:
|
2. Usage:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
@@ -309,6 +360,9 @@ python run.py --execution-provider openvino
|
|||||||
- Use a screen capture tool like OBS to stream.
|
- Use a screen capture tool like OBS to stream.
|
||||||
- To change the face, select a new source image.
|
- To change the face, select a new source image.
|
||||||
|
|
||||||
|
## Download all models in this huggingface link
|
||||||
|
- [**Download models here**](https://huggingface.co/hacksider/deep-live-cam/tree/main)
|
||||||
|
|
||||||
## Command Line Arguments (Unmaintained)
|
## Command Line Arguments (Unmaintained)
|
||||||
|
|
||||||
```
|
```
|
||||||
@@ -362,6 +416,7 @@ Looking for a CLI mode? Using the -s/--source argument will make the run program
|
|||||||
- [kier007](https://github.com/kier007): for improving the user experience
|
- [kier007](https://github.com/kier007): for improving the user experience
|
||||||
- [qitianai](https://github.com/qitianai): for multi-lingual support
|
- [qitianai](https://github.com/qitianai): for multi-lingual support
|
||||||
- [laurigates](https://github.com/laurigates): Decoupling stuffs to make everything faster!
|
- [laurigates](https://github.com/laurigates): Decoupling stuffs to make everything faster!
|
||||||
|
- [maxwbuckley](https://github.com/maxwbuckley): For making the effort to optimize this for mac!
|
||||||
- and [all developers](https://github.com/hacksider/Deep-Live-Cam/graphs/contributors) behind libraries used in this project.
|
- and [all developers](https://github.com/hacksider/Deep-Live-Cam/graphs/contributors) behind libraries used in this project.
|
||||||
- Footnote: Please be informed that the base author of the code is [s0md3v](https://github.com/s0md3v/roop)
|
- Footnote: Please be informed that the base author of the code is [s0md3v](https://github.com/s0md3v/roop)
|
||||||
- All the wonderful users who helped make this project go viral by starring the repo ❤️
|
- All the wonderful users who helped make this project go viral by starring the repo ❤️
|
||||||
|
|||||||
@@ -0,0 +1,181 @@
|
|||||||
|
"""Standalone pipeline benchmark — no UI required.
|
||||||
|
|
||||||
|
Captures 200 frames from the webcam and runs the full face swap pipeline,
|
||||||
|
printing per-stage timing and effective FPS.
|
||||||
|
"""
|
||||||
|
import os, sys, time, cv2, numpy as np, queue, threading
|
||||||
|
|
||||||
|
# PATH fix for cuDNN (Windows only)
|
||||||
|
if sys.platform == "win32":
|
||||||
|
_sp = os.path.join(sys.prefix, "Lib", "site-packages")
|
||||||
|
_torch_lib = os.path.join(_sp, "torch", "lib")
|
||||||
|
if os.path.isdir(_torch_lib):
|
||||||
|
os.environ["PATH"] = _torch_lib + os.pathsep + os.environ["PATH"]
|
||||||
|
|
||||||
|
import insightface
|
||||||
|
from insightface.app import FaceAnalysis
|
||||||
|
from modules.processors.frame.face_swapper import _fast_paste_back
|
||||||
|
from modules import platform_info
|
||||||
|
|
||||||
|
platform_info.print_banner()
|
||||||
|
|
||||||
|
# Pick providers based on what's actually available on this machine.
|
||||||
|
if platform_info.HAS_CUDA_PROVIDER:
|
||||||
|
_providers = ["CUDAExecutionProvider", "CPUExecutionProvider"]
|
||||||
|
elif platform_info.HAS_COREML_PROVIDER:
|
||||||
|
_providers = ["CoreMLExecutionProvider", "CPUExecutionProvider"]
|
||||||
|
else:
|
||||||
|
_providers = ["CPUExecutionProvider"]
|
||||||
|
|
||||||
|
# --- Init models (same as the app) ---
|
||||||
|
print(f"Loading models with providers={_providers}...")
|
||||||
|
fa = FaceAnalysis(
|
||||||
|
name="buffalo_l",
|
||||||
|
providers=_providers,
|
||||||
|
allowed_modules=["detection", "recognition", "landmark_2d_106"],
|
||||||
|
)
|
||||||
|
fa.prepare(ctx_id=0, det_size=(640, 640))
|
||||||
|
swap_model = insightface.model_zoo.get_model(
|
||||||
|
"models/inswapper_128.onnx",
|
||||||
|
providers=_providers,
|
||||||
|
)
|
||||||
|
face_size = swap_model.input_size[0]
|
||||||
|
aimg_dummy = np.empty((face_size, face_size, 3), dtype=np.uint8)
|
||||||
|
|
||||||
|
# --- Camera setup ---
|
||||||
|
# Windows: DirectShow explicit for MJPEG 1080p60 support.
|
||||||
|
# macOS/Linux: default backend (AVFoundation / V4L2).
|
||||||
|
print("Opening camera at 1080p60 MJPEG...")
|
||||||
|
if sys.platform == "win32":
|
||||||
|
cap = cv2.VideoCapture(0, cv2.CAP_DSHOW)
|
||||||
|
else:
|
||||||
|
cap = cv2.VideoCapture(0)
|
||||||
|
cap.set(cv2.CAP_PROP_FOURCC, cv2.VideoWriter_fourcc(*"MJPG"))
|
||||||
|
cap.set(cv2.CAP_PROP_FRAME_WIDTH, 1920)
|
||||||
|
cap.set(cv2.CAP_PROP_FRAME_HEIGHT, 1080)
|
||||||
|
cap.set(cv2.CAP_PROP_FPS, 60)
|
||||||
|
time.sleep(0.5)
|
||||||
|
|
||||||
|
# Warmup + get source face
|
||||||
|
for _ in range(15):
|
||||||
|
cap.read()
|
||||||
|
ret, src_frame = cap.read()
|
||||||
|
faces = fa.get(src_frame)
|
||||||
|
if not faces:
|
||||||
|
print("ERROR: No face detected in warmup frame")
|
||||||
|
cap.release()
|
||||||
|
sys.exit(1)
|
||||||
|
source_face = faces[0]
|
||||||
|
print(f"Source face acquired. Frame: {src_frame.shape}")
|
||||||
|
|
||||||
|
# --- Capture thread (same as app) ---
|
||||||
|
capture_queue = queue.Queue(maxsize=2)
|
||||||
|
stop_event = threading.Event()
|
||||||
|
|
||||||
|
def capture_thread():
|
||||||
|
while not stop_event.is_set():
|
||||||
|
ret, frame = cap.read()
|
||||||
|
if not ret:
|
||||||
|
break
|
||||||
|
try:
|
||||||
|
capture_queue.put_nowait(frame)
|
||||||
|
except queue.Full:
|
||||||
|
try:
|
||||||
|
capture_queue.get_nowait()
|
||||||
|
except queue.Empty:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
capture_queue.put_nowait(frame)
|
||||||
|
except queue.Full:
|
||||||
|
pass
|
||||||
|
|
||||||
|
cap_t = threading.Thread(target=capture_thread, daemon=True)
|
||||||
|
cap_t.start()
|
||||||
|
|
||||||
|
# --- Warmup processing ---
|
||||||
|
print("Warming up pipeline...")
|
||||||
|
for _ in range(20):
|
||||||
|
try:
|
||||||
|
frame = capture_queue.get(timeout=0.1)
|
||||||
|
except queue.Empty:
|
||||||
|
continue
|
||||||
|
f = frame.copy()
|
||||||
|
det_faces = fa.get(f)
|
||||||
|
if det_faces:
|
||||||
|
tgt = min(det_faces, key=lambda x: x.bbox[0])
|
||||||
|
bgr_fake, M = swap_model.get(f, tgt, source_face, paste_back=False)
|
||||||
|
_fast_paste_back(f, bgr_fake, aimg_dummy, M)
|
||||||
|
|
||||||
|
# --- Benchmark ---
|
||||||
|
N = 200
|
||||||
|
print(f"\nBenchmarking {N} frames...")
|
||||||
|
|
||||||
|
t_queue, t_det, t_onnx, t_paste, t_copy, t_cvt, t_total = [], [], [], [], [], [], []
|
||||||
|
det_count = 0
|
||||||
|
cached_face = None
|
||||||
|
|
||||||
|
for i in range(N):
|
||||||
|
tt = time.perf_counter()
|
||||||
|
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
try:
|
||||||
|
frame = capture_queue.get(timeout=0.1)
|
||||||
|
except queue.Empty:
|
||||||
|
continue
|
||||||
|
t_queue.append((time.perf_counter() - t0) * 1000)
|
||||||
|
|
||||||
|
# Detection every 3rd frame — det-only (no landmark/recognition)
|
||||||
|
det_count += 1
|
||||||
|
if det_count % 3 == 0:
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
from insightface.app.common import Face as _Face
|
||||||
|
bboxes, kpss = fa.det_model.detect(frame, max_num=0, metric='default')
|
||||||
|
if bboxes.shape[0] > 0:
|
||||||
|
idx = int(bboxes[:, 0].argmin())
|
||||||
|
cached_face = _Face(bbox=bboxes[idx, :4], kps=kpss[idx], det_score=bboxes[idx, 4])
|
||||||
|
t_det.append((time.perf_counter() - t0) * 1000)
|
||||||
|
|
||||||
|
if cached_face is not None:
|
||||||
|
# No frame.copy() — _fast_paste_back writes in-place, we own the frame
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
bgr_fake, M = swap_model.get(frame, cached_face, source_face, paste_back=False)
|
||||||
|
t_onnx.append((time.perf_counter() - t0) * 1000)
|
||||||
|
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
result = _fast_paste_back(frame, bgr_fake, aimg_dummy, M)
|
||||||
|
t_paste.append((time.perf_counter() - t0) * 1000)
|
||||||
|
|
||||||
|
# Display prep — resize then flip (no cvtColor needed)
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
small = cv2.resize(result, (640, 360))
|
||||||
|
_ = small[:, :, ::-1] # BGR→RGB zero-copy
|
||||||
|
t_cvt.append((time.perf_counter() - t0) * 1000)
|
||||||
|
|
||||||
|
t_total.append((time.perf_counter() - tt) * 1000)
|
||||||
|
|
||||||
|
stop_event.set()
|
||||||
|
cap.release()
|
||||||
|
|
||||||
|
# --- Results ---
|
||||||
|
def s(name, arr):
|
||||||
|
if not arr:
|
||||||
|
return
|
||||||
|
avg = sum(arr) / len(arr)
|
||||||
|
print(f" {name:25s}: avg={avg:6.1f}ms min={min(arr):5.1f}ms max={max(arr):6.1f}ms n={len(arr)}")
|
||||||
|
|
||||||
|
print(f"\n{'='*55}")
|
||||||
|
print(f" 1080p Pipeline Benchmark ({len(t_total)} frames)")
|
||||||
|
print(f"{'='*55}")
|
||||||
|
s("queue.get (wait for cam)", t_queue)
|
||||||
|
s("detection (fa.get)", t_det)
|
||||||
|
s("frame.copy()", t_copy)
|
||||||
|
s("ONNX swap", t_onnx)
|
||||||
|
s("_fast_paste_back", t_paste)
|
||||||
|
s("cvtColor BGR->RGB", t_cvt)
|
||||||
|
s("TOTAL per frame", t_total)
|
||||||
|
|
||||||
|
avg_total = sum(t_total) / len(t_total)
|
||||||
|
avg_queue = sum(t_queue) / len(t_queue)
|
||||||
|
print(f"\n Effective FPS: {1000/avg_total:.1f}")
|
||||||
|
print(f" FPS (excl. cam wait): {1000/(avg_total - avg_queue):.1f}")
|
||||||
|
print(f"{'='*55}")
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
just put the models in this folder -
|
just put the models in this folder -
|
||||||
|
|
||||||
https://huggingface.co/hacksider/deep-live-cam/resolve/main/inswapper_128_fp16.onnx?download=true
|
https://huggingface.co/hacksider/deep-live-cam/resolve/main/inswapper_128_fp16.onnx?download=true
|
||||||
https://github.com/TencentARC/GFPGAN/releases/download/v1.3.4/GFPGANv1.4.pth
|
https://huggingface.co/hacksider/deep-live-cam/resolve/main/gfpgan-1024.onnx?download=true
|
||||||
|
|||||||
+38
-18
@@ -1,18 +1,38 @@
|
|||||||
import os
|
import os
|
||||||
import cv2
|
import cv2
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
||||||
# Utility function to support unicode characters in file paths for reading
|
|
||||||
def imread_unicode(path, flags=cv2.IMREAD_COLOR):
|
# Utility function to support unicode characters in file paths for reading.
|
||||||
return cv2.imdecode(np.fromfile(path, dtype=np.uint8), flags)
|
# OpenCV's cv2.imread() encodes the path with the locale ANSI code page on
|
||||||
|
# Windows, so it silently returns None for paths containing non-ASCII
|
||||||
# Utility function to support unicode characters in file paths for writing
|
# characters (Chinese, Japanese, Cyrillic, accents, ...). Reading the bytes
|
||||||
def imwrite_unicode(path, img, params=None):
|
# through NumPy (which uses Python's unicode-aware file I/O) and decoding them
|
||||||
root, ext = os.path.splitext(path)
|
# in memory sidesteps that limitation. Returns None on failure, matching
|
||||||
if not ext:
|
# cv2.imread() so it stays a drop-in replacement.
|
||||||
ext = ".png"
|
def imread_unicode(path, flags=cv2.IMREAD_COLOR):
|
||||||
result, encoded_img = cv2.imencode(ext, img, params if params else [])
|
try:
|
||||||
result, encoded_img = cv2.imencode(f".{ext}", img, params if params is not None else [])
|
data = np.fromfile(path, dtype=np.uint8)
|
||||||
encoded_img.tofile(path)
|
if data.size == 0:
|
||||||
return True
|
return None
|
||||||
return False
|
return cv2.imdecode(data, flags)
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
# Utility function to support unicode characters in file paths for writing.
|
||||||
|
# cv2.imwrite() has the same ANSI-path limitation, so we encode the image in
|
||||||
|
# memory and write the bytes out with NumPy's unicode-aware file I/O. Returns
|
||||||
|
# True/False like cv2.imwrite() so it stays a drop-in replacement.
|
||||||
|
def imwrite_unicode(path, img, params=None):
|
||||||
|
try:
|
||||||
|
root, ext = os.path.splitext(path)
|
||||||
|
if not ext:
|
||||||
|
ext = ".png"
|
||||||
|
result, encoded_img = cv2.imencode(ext, img, params if params is not None else [])
|
||||||
|
if not result:
|
||||||
|
return False
|
||||||
|
encoded_img.tofile(path)
|
||||||
|
return True
|
||||||
|
except Exception:
|
||||||
|
return False
|
||||||
|
|||||||
+7
-2
@@ -14,8 +14,13 @@ def get_video_frame(video_path: str, frame_number: int = 0) -> Any:
|
|||||||
if modules.globals.color_correction:
|
if modules.globals.color_correction:
|
||||||
capture.set(cv2.CAP_PROP_CONVERT_RGB, 1)
|
capture.set(cv2.CAP_PROP_CONVERT_RGB, 1)
|
||||||
|
|
||||||
frame_total = capture.get(cv2.CAP_PROP_FRAME_COUNT)
|
frame_total = int(capture.get(cv2.CAP_PROP_FRAME_COUNT))
|
||||||
capture.set(cv2.CAP_PROP_POS_FRAMES, min(frame_total, frame_number - 1))
|
if frame_total <= 0:
|
||||||
|
capture.release()
|
||||||
|
return None
|
||||||
|
|
||||||
|
target_index = 0 if frame_number <= 1 else min(frame_total - 1, frame_number - 1)
|
||||||
|
capture.set(cv2.CAP_PROP_POS_FRAMES, target_index)
|
||||||
has_frame, frame = capture.read()
|
has_frame, frame = capture.read()
|
||||||
|
|
||||||
if has_frame and modules.globals.color_correction:
|
if has_frame and modules.globals.color_correction:
|
||||||
|
|||||||
@@ -1,10 +1,21 @@
|
|||||||
import numpy as np
|
import numpy as np
|
||||||
from sklearn.cluster import KMeans
|
from sklearn.cluster import KMeans
|
||||||
from sklearn.metrics import silhouette_score
|
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
def find_cluster_centroids(embeddings, max_k=10) -> Any:
|
def find_cluster_centroids(embeddings, max_k=10) -> Any:
|
||||||
|
n_samples = len(embeddings)
|
||||||
|
if n_samples == 0:
|
||||||
|
raise ValueError("embeddings must not be empty")
|
||||||
|
if max_k < 1:
|
||||||
|
raise ValueError("max_k must be at least 1")
|
||||||
|
|
||||||
|
max_k = min(max_k, n_samples)
|
||||||
|
if max_k == 1:
|
||||||
|
kmeans = KMeans(n_clusters=1, random_state=0)
|
||||||
|
kmeans.fit(embeddings)
|
||||||
|
return kmeans.cluster_centers_
|
||||||
|
|
||||||
inertia = []
|
inertia = []
|
||||||
cluster_centroids = []
|
cluster_centroids = []
|
||||||
K = range(1, max_k+1)
|
K = range(1, max_k+1)
|
||||||
|
|||||||
+102
-46
@@ -2,7 +2,7 @@ import os
|
|||||||
import sys
|
import sys
|
||||||
# single thread doubles cuda performance - needs to be set before torch import
|
# single thread doubles cuda performance - needs to be set before torch import
|
||||||
if any(arg.startswith('--execution-provider') for arg in sys.argv):
|
if any(arg.startswith('--execution-provider') for arg in sys.argv):
|
||||||
os.environ['OMP_NUM_THREADS'] = '1'
|
os.environ['OMP_NUM_THREADS'] = '6'
|
||||||
# reduce tensorflow log level
|
# reduce tensorflow log level
|
||||||
os.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'
|
os.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'
|
||||||
import warnings
|
import warnings
|
||||||
@@ -17,12 +17,16 @@ try:
|
|||||||
except ImportError:
|
except ImportError:
|
||||||
HAS_TORCH = False
|
HAS_TORCH = False
|
||||||
import onnxruntime
|
import onnxruntime
|
||||||
import tensorflow
|
try:
|
||||||
|
import tensorflow
|
||||||
|
HAS_TENSORFLOW = True
|
||||||
|
except ImportError:
|
||||||
|
HAS_TENSORFLOW = False
|
||||||
|
|
||||||
import modules.globals
|
import modules.globals
|
||||||
import modules.metadata
|
import modules.metadata
|
||||||
import modules.ui as ui
|
import modules.ui as ui
|
||||||
from modules.processors.frame.core import get_frame_processors_modules
|
from modules.processors.frame.core import get_frame_processors_modules, process_video_in_memory
|
||||||
from modules.utilities import has_image_extension, is_image, is_video, detect_fps, create_video, extract_frames, get_temp_frame_paths, restore_audio, create_temp, move_temp, clean_temp, normalize_output_path
|
from modules.utilities import has_image_extension, is_image, is_video, detect_fps, create_video, extract_frames, get_temp_frame_paths, restore_audio, create_temp, move_temp, clean_temp, normalize_output_path
|
||||||
|
|
||||||
if HAS_TORCH and 'ROCMExecutionProvider' in modules.globals.execution_providers:
|
if HAS_TORCH and 'ROCMExecutionProvider' in modules.globals.execution_providers:
|
||||||
@@ -53,8 +57,8 @@ def parse_args() -> None:
|
|||||||
program.add_argument('--live-mirror', help='The live camera display as you see it in the front-facing camera frame', dest='live_mirror', action='store_true', default=False)
|
program.add_argument('--live-mirror', help='The live camera display as you see it in the front-facing camera frame', dest='live_mirror', action='store_true', default=False)
|
||||||
program.add_argument('--live-resizable', help='The live camera frame is resizable', dest='live_resizable', action='store_true', default=False)
|
program.add_argument('--live-resizable', help='The live camera frame is resizable', dest='live_resizable', action='store_true', default=False)
|
||||||
program.add_argument('--max-memory', help='maximum amount of RAM in GB', dest='max_memory', type=int, default=suggest_max_memory())
|
program.add_argument('--max-memory', help='maximum amount of RAM in GB', dest='max_memory', type=int, default=suggest_max_memory())
|
||||||
program.add_argument('--execution-provider', help='execution provider', dest='execution_provider', default=['cpu'], choices=suggest_execution_providers(), nargs='+')
|
program.add_argument('--execution-provider', help='execution provider', dest='execution_provider', default=[suggest_default_execution_provider()], choices=suggest_execution_providers(), nargs='+')
|
||||||
program.add_argument('--execution-threads', help='number of execution threads', dest='execution_threads', type=int, default=suggest_execution_threads())
|
program.add_argument('--execution-threads', help='number of execution threads', dest='execution_threads', type=int, default=None)
|
||||||
program.add_argument('-v', '--version', action='version', version=f'{modules.metadata.name} {modules.metadata.version}')
|
program.add_argument('-v', '--version', action='version', version=f'{modules.metadata.name} {modules.metadata.version}')
|
||||||
|
|
||||||
# register deprecated args
|
# register deprecated args
|
||||||
@@ -86,6 +90,12 @@ def parse_args() -> None:
|
|||||||
modules.globals.execution_threads = args.execution_threads
|
modules.globals.execution_threads = args.execution_threads
|
||||||
modules.globals.lang = args.lang
|
modules.globals.lang = args.lang
|
||||||
|
|
||||||
|
# The argparse default (None) avoids evaluating suggest_execution_threads()
|
||||||
|
# before providers are decoded, and deprecated-arg overrides above may
|
||||||
|
# have already set execution_threads.
|
||||||
|
if modules.globals.execution_threads is None:
|
||||||
|
modules.globals.execution_threads = suggest_execution_threads()
|
||||||
|
|
||||||
#for ENHANCER tumblers:
|
#for ENHANCER tumblers:
|
||||||
for enhancer_key in ('face_enhancer', 'face_enhancer_gpen256', 'face_enhancer_gpen512'):
|
for enhancer_key in ('face_enhancer', 'face_enhancer_gpen256', 'face_enhancer_gpen512'):
|
||||||
modules.globals.fp_ui[enhancer_key] = enhancer_key in args.frame_processor
|
modules.globals.fp_ui[enhancer_key] = enhancer_key in args.frame_processor
|
||||||
@@ -127,6 +137,15 @@ def suggest_max_memory() -> int:
|
|||||||
return 16
|
return 16
|
||||||
|
|
||||||
|
|
||||||
|
def suggest_default_execution_provider() -> str:
|
||||||
|
"""Pick the best available provider: cuda > rocm > coreml > openvino > dml > cpu."""
|
||||||
|
available = encode_execution_providers(onnxruntime.get_available_providers())
|
||||||
|
for pref in ('cuda', 'rocm', 'coreml', 'openvino', 'dml'):
|
||||||
|
if pref in available:
|
||||||
|
return pref
|
||||||
|
return 'cpu'
|
||||||
|
|
||||||
|
|
||||||
def suggest_execution_providers() -> List[str]:
|
def suggest_execution_providers() -> List[str]:
|
||||||
return encode_execution_providers(onnxruntime.get_available_providers())
|
return encode_execution_providers(onnxruntime.get_available_providers())
|
||||||
|
|
||||||
@@ -143,8 +162,9 @@ def suggest_execution_threads() -> int:
|
|||||||
if 'ROCMExecutionProvider' in modules.globals.execution_providers:
|
if 'ROCMExecutionProvider' in modules.globals.execution_providers:
|
||||||
return 1
|
return 1
|
||||||
if 'CUDAExecutionProvider' in modules.globals.execution_providers:
|
if 'CUDAExecutionProvider' in modules.globals.execution_providers:
|
||||||
# For CUDA, use more threads for parallel frame processing
|
return 2
|
||||||
return min(cpu_count, 16)
|
if 'OpenVINOExecutionProvider' in modules.globals.execution_providers:
|
||||||
|
return 1
|
||||||
|
|
||||||
# For CPU execution, use most cores but leave some for system
|
# For CPU execution, use most cores but leave some for system
|
||||||
return max(4, min(cpu_count - 2, 16))
|
return max(4, min(cpu_count - 2, 16))
|
||||||
@@ -152,14 +172,17 @@ def suggest_execution_threads() -> int:
|
|||||||
|
|
||||||
def limit_resources() -> None:
|
def limit_resources() -> None:
|
||||||
# prevent tensorflow memory leak
|
# prevent tensorflow memory leak
|
||||||
gpus = tensorflow.config.experimental.list_physical_devices('GPU')
|
if HAS_TENSORFLOW:
|
||||||
for gpu in gpus:
|
gpus = tensorflow.config.experimental.list_physical_devices('GPU')
|
||||||
tensorflow.config.experimental.set_memory_growth(gpu, True)
|
for gpu in gpus:
|
||||||
|
tensorflow.config.experimental.set_memory_growth(gpu, True)
|
||||||
# limit memory usage
|
# limit memory usage
|
||||||
if modules.globals.max_memory:
|
if modules.globals.max_memory:
|
||||||
memory = modules.globals.max_memory * 1024 ** 3
|
# setrlimit(RLIMIT_DATA) fails with EINVAL on macOS, crashing on launch.
|
||||||
|
# See https://github.com/hacksider/Deep-Live-Cam/issues/1848
|
||||||
if platform.system().lower() == 'darwin':
|
if platform.system().lower() == 'darwin':
|
||||||
memory = modules.globals.max_memory * 1024 ** 6
|
return
|
||||||
|
memory = modules.globals.max_memory * 1024 ** 3
|
||||||
if platform.system().lower() == 'windows':
|
if platform.system().lower() == 'windows':
|
||||||
import ctypes
|
import ctypes
|
||||||
kernel32 = ctypes.windll.kernel32
|
kernel32 = ctypes.windll.kernel32
|
||||||
@@ -223,40 +246,69 @@ def start() -> None:
|
|||||||
if modules.globals.nsfw_filter and ui.check_and_ignore_nsfw(modules.globals.target_path, destroy):
|
if modules.globals.nsfw_filter and ui.check_and_ignore_nsfw(modules.globals.target_path, destroy):
|
||||||
return
|
return
|
||||||
|
|
||||||
extraction_start = time.time()
|
# Detect FPS early (needed by both pipelines)
|
||||||
if not modules.globals.map_faces:
|
|
||||||
update_status('Creating temp resources...')
|
|
||||||
create_temp(modules.globals.target_path)
|
|
||||||
update_status('Extracting frames...')
|
|
||||||
extract_frames(modules.globals.target_path)
|
|
||||||
extraction_time = time.time() - extraction_start
|
|
||||||
update_status(f'Frame extraction completed in {extraction_time:.2f}s')
|
|
||||||
|
|
||||||
temp_frame_paths = get_temp_frame_paths(modules.globals.target_path)
|
|
||||||
total_frames = len(temp_frame_paths)
|
|
||||||
update_status(f'Processing {total_frames} frames with {modules.globals.execution_threads} threads...')
|
|
||||||
|
|
||||||
processing_start = time.time()
|
|
||||||
for frame_processor in get_frame_processors_modules(modules.globals.frame_processors):
|
|
||||||
update_status('Progressing...', frame_processor.NAME)
|
|
||||||
frame_processor.process_video(modules.globals.source_path, temp_frame_paths)
|
|
||||||
release_resources()
|
|
||||||
processing_time = time.time() - processing_start
|
|
||||||
fps_processing = total_frames / processing_time if processing_time > 0 else 0
|
|
||||||
update_status(f'Frame processing completed in {processing_time:.2f}s ({fps_processing:.2f} fps)')
|
|
||||||
|
|
||||||
# handles fps
|
|
||||||
encoding_start = time.time()
|
|
||||||
if modules.globals.keep_fps:
|
if modules.globals.keep_fps:
|
||||||
update_status('Detecting fps...')
|
update_status('Detecting fps...')
|
||||||
fps = detect_fps(modules.globals.target_path)
|
fps = detect_fps(modules.globals.target_path)
|
||||||
update_status(f'Creating video with {fps} fps...')
|
|
||||||
create_video(modules.globals.target_path, fps)
|
|
||||||
else:
|
else:
|
||||||
update_status('Creating video with 30.0 fps...')
|
fps = 30.0
|
||||||
create_video(modules.globals.target_path)
|
|
||||||
encoding_time = time.time() - encoding_start
|
video_created = False
|
||||||
update_status(f'Video encoding completed in {encoding_time:.2f}s')
|
|
||||||
|
# --- In-memory pipeline (non-map_faces only) ---
|
||||||
|
# Reads frames from FFmpeg pipe, processes in memory, encodes directly.
|
||||||
|
# Eliminates all per-frame PNG disk I/O for a major speed-up.
|
||||||
|
if not modules.globals.map_faces:
|
||||||
|
update_status(f'Processing video in-memory at {fps} fps...')
|
||||||
|
create_temp(modules.globals.target_path)
|
||||||
|
|
||||||
|
processing_start = time.time()
|
||||||
|
video_created = process_video_in_memory(
|
||||||
|
modules.globals.source_path,
|
||||||
|
modules.globals.target_path,
|
||||||
|
fps,
|
||||||
|
)
|
||||||
|
processing_time = time.time() - processing_start
|
||||||
|
release_resources()
|
||||||
|
|
||||||
|
if video_created:
|
||||||
|
update_status(f'In-memory processing + encoding completed in {processing_time:.2f}s')
|
||||||
|
|
||||||
|
# --- Disk-based fallback (required for map_faces, or if pipe failed) ---
|
||||||
|
if not video_created:
|
||||||
|
if not modules.globals.map_faces:
|
||||||
|
update_status('Falling back to disk-based processing...')
|
||||||
|
|
||||||
|
extraction_start = time.time()
|
||||||
|
create_temp(modules.globals.target_path)
|
||||||
|
update_status('Extracting frames...')
|
||||||
|
extract_frames(modules.globals.target_path)
|
||||||
|
extraction_time = time.time() - extraction_start
|
||||||
|
|
||||||
|
temp_frame_paths = get_temp_frame_paths(modules.globals.target_path)
|
||||||
|
total_frames = len(temp_frame_paths)
|
||||||
|
update_status(f'Processing {total_frames} frames with {modules.globals.execution_threads} threads...')
|
||||||
|
|
||||||
|
processing_start = time.time()
|
||||||
|
for frame_processor in get_frame_processors_modules(modules.globals.frame_processors):
|
||||||
|
update_status('Progressing...', frame_processor.NAME)
|
||||||
|
frame_processor.process_video(modules.globals.source_path, temp_frame_paths)
|
||||||
|
release_resources()
|
||||||
|
processing_time = time.time() - processing_start
|
||||||
|
fps_processing = total_frames / processing_time if processing_time > 0 else 0
|
||||||
|
update_status(f'Frame processing completed in {processing_time:.2f}s ({fps_processing:.2f} fps)')
|
||||||
|
|
||||||
|
encoding_start = time.time()
|
||||||
|
update_status(f'Creating video with {fps} fps...')
|
||||||
|
video_created = create_video(modules.globals.target_path, fps)
|
||||||
|
encoding_time = time.time() - encoding_start
|
||||||
|
if video_created:
|
||||||
|
update_status(f'Video encoding completed in {encoding_time:.2f}s')
|
||||||
|
|
||||||
|
if not video_created:
|
||||||
|
update_status('Video encoding failed. No temporary output video was created.')
|
||||||
|
clean_temp(modules.globals.target_path)
|
||||||
|
return
|
||||||
|
|
||||||
# handle audio
|
# handle audio
|
||||||
if modules.globals.keep_audio:
|
if modules.globals.keep_audio:
|
||||||
@@ -272,8 +324,8 @@ def start() -> None:
|
|||||||
clean_temp(modules.globals.target_path)
|
clean_temp(modules.globals.target_path)
|
||||||
|
|
||||||
total_time = time.time() - start_time
|
total_time = time.time() - start_time
|
||||||
if is_video(modules.globals.target_path):
|
if is_video(modules.globals.target_path) and modules.globals.output_path and os.path.isfile(modules.globals.output_path):
|
||||||
update_status(f'Processing to video succeed! Total time: {total_time:.2f}s')
|
update_status(f'Video processing succeeded! Total time: {total_time:.2f}s')
|
||||||
else:
|
else:
|
||||||
update_status('Processing to video failed!')
|
update_status('Processing to video failed!')
|
||||||
|
|
||||||
@@ -281,7 +333,8 @@ def start() -> None:
|
|||||||
def destroy(to_quit=True) -> None:
|
def destroy(to_quit=True) -> None:
|
||||||
if modules.globals.target_path:
|
if modules.globals.target_path:
|
||||||
clean_temp(modules.globals.target_path)
|
clean_temp(modules.globals.target_path)
|
||||||
if to_quit: quit()
|
if to_quit:
|
||||||
|
quit()
|
||||||
|
|
||||||
|
|
||||||
def run() -> None:
|
def run() -> None:
|
||||||
@@ -291,9 +344,12 @@ def run() -> None:
|
|||||||
for frame_processor in get_frame_processors_modules(modules.globals.frame_processors):
|
for frame_processor in get_frame_processors_modules(modules.globals.frame_processors):
|
||||||
if not frame_processor.pre_check():
|
if not frame_processor.pre_check():
|
||||||
return
|
return
|
||||||
|
# Pre-load face analyser in main thread before GUI starts
|
||||||
|
#from modules.face_analyser import get_face_analyser
|
||||||
|
#get_face_analyser()
|
||||||
limit_resources()
|
limit_resources()
|
||||||
if modules.globals.headless:
|
if modules.globals.headless:
|
||||||
start()
|
start()
|
||||||
else:
|
else:
|
||||||
window = ui.init(start, destroy, modules.globals.lang)
|
window = ui.init(start, destroy, modules.globals.lang)
|
||||||
window.mainloop()
|
window.mainloop()
|
||||||
+193
-15
@@ -4,9 +4,8 @@ from typing import Any
|
|||||||
import insightface
|
import insightface
|
||||||
import threading
|
import threading
|
||||||
|
|
||||||
import cv2
|
|
||||||
import numpy as np
|
|
||||||
import modules.globals
|
import modules.globals
|
||||||
|
from modules import imread_unicode, imwrite_unicode
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
from modules.typing import Frame
|
from modules.typing import Frame
|
||||||
from modules.cluster_analysis import find_cluster_centroids, find_closest_centroid
|
from modules.cluster_analysis import find_cluster_centroids, find_closest_centroid
|
||||||
@@ -16,6 +15,8 @@ from pathlib import Path
|
|||||||
FACE_ANALYSER = None
|
FACE_ANALYSER = None
|
||||||
FACE_ANALYSER_LOCK = threading.Lock()
|
FACE_ANALYSER_LOCK = threading.Lock()
|
||||||
|
|
||||||
|
DET_SIZE = (640, 640)
|
||||||
|
|
||||||
|
|
||||||
def get_face_analyser() -> Any:
|
def get_face_analyser() -> Any:
|
||||||
"""Get face analyser with thread-safe initialization."""
|
"""Get face analyser with thread-safe initialization."""
|
||||||
@@ -25,29 +26,199 @@ def get_face_analyser() -> Any:
|
|||||||
with FACE_ANALYSER_LOCK:
|
with FACE_ANALYSER_LOCK:
|
||||||
# Double-check after acquiring lock
|
# Double-check after acquiring lock
|
||||||
if FACE_ANALYSER is None:
|
if FACE_ANALYSER is None:
|
||||||
|
from modules.processors.frame._onnx_enhancer import (
|
||||||
|
build_provider_config,
|
||||||
|
)
|
||||||
|
from modules.model_downloader import ensure_insightface_pack
|
||||||
|
|
||||||
|
ensure_insightface_pack('buffalo_l')
|
||||||
|
providers = build_provider_config()
|
||||||
FACE_ANALYSER = insightface.app.FaceAnalysis(
|
FACE_ANALYSER = insightface.app.FaceAnalysis(
|
||||||
name='buffalo_l',
|
name='buffalo_l',
|
||||||
providers=modules.globals.execution_providers,
|
providers=providers,
|
||||||
allowed_modules=['detection', 'recognition']
|
allowed_modules=['detection', 'recognition', 'landmark_2d_106']
|
||||||
)
|
)
|
||||||
FACE_ANALYSER.prepare(ctx_id=0, det_size=(320, 320))
|
FACE_ANALYSER.prepare(ctx_id=0, det_size=DET_SIZE)
|
||||||
|
_optimize_det_model(FACE_ANALYSER, providers)
|
||||||
return FACE_ANALYSER
|
return FACE_ANALYSER
|
||||||
|
|
||||||
|
|
||||||
def get_one_face(frame: Frame) -> Any:
|
def _optimize_det_model(fa: Any, providers) -> None:
|
||||||
face = get_face_analyser().get(frame)
|
"""Replace the detection model's ONNX session with a CoreML-optimized one.
|
||||||
|
|
||||||
|
Folds dynamic Shape→Gather chains into constants (the input size is
|
||||||
|
fixed at det_size), eliminating CPU↔ANE partition boundaries in the
|
||||||
|
RetinaFace FPN upsampling path. 21ms → 4ms on M3 Max.
|
||||||
|
"""
|
||||||
|
from modules.onnx_optimize import optimize_for_coreml, IS_APPLE_SILICON
|
||||||
|
if not IS_APPLE_SILICON:
|
||||||
|
return
|
||||||
|
|
||||||
|
det_model = fa.det_model
|
||||||
|
model_path = getattr(det_model, 'model_file', None)
|
||||||
|
if model_path is None or not os.path.exists(model_path):
|
||||||
|
return
|
||||||
|
|
||||||
|
input_shape = (1, 3, DET_SIZE[1], DET_SIZE[0])
|
||||||
|
optimized_path = optimize_for_coreml(model_path, input_shape=input_shape)
|
||||||
|
if optimized_path == model_path:
|
||||||
|
return
|
||||||
|
|
||||||
|
import onnxruntime
|
||||||
|
session_options = onnxruntime.SessionOptions()
|
||||||
|
session_options.graph_optimization_level = (
|
||||||
|
onnxruntime.GraphOptimizationLevel.ORT_ENABLE_ALL
|
||||||
|
)
|
||||||
|
|
||||||
|
# Route detection to GPU shader cores (CPUAndGPU) instead of ANE.
|
||||||
|
# This lets detection run concurrently with the swap model on the
|
||||||
|
# ANE, overlapping the two inference calls. Detection is fast
|
||||||
|
# enough on GPU (~4ms) and this frees ANE for the heavier swap.
|
||||||
|
det_providers = []
|
||||||
|
for p in providers:
|
||||||
|
name = p[0] if isinstance(p, tuple) else p
|
||||||
|
if name == "CoreMLExecutionProvider":
|
||||||
|
det_providers.append((
|
||||||
|
"CoreMLExecutionProvider",
|
||||||
|
{"ModelFormat": "MLProgram", "MLComputeUnits": "CPUAndGPU"},
|
||||||
|
))
|
||||||
|
else:
|
||||||
|
det_providers.append(p)
|
||||||
|
|
||||||
|
det_model.session = onnxruntime.InferenceSession(
|
||||||
|
optimized_path, sess_options=session_options, providers=det_providers,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _needs_landmark() -> bool:
|
||||||
|
"""Check whether any active feature requires 106-point landmarks.
|
||||||
|
|
||||||
|
Landmarks are needed by face enhancers and mouth masking, but not
|
||||||
|
by the face swapper alone.
|
||||||
|
"""
|
||||||
|
if getattr(modules.globals, "mouth_mask", False):
|
||||||
|
return True
|
||||||
|
processors = getattr(modules.globals, "frame_processors", [])
|
||||||
|
return any(p in processors for p in
|
||||||
|
("face_enhancer", "face_enhancer_gpen256", "face_enhancer_gpen512"))
|
||||||
|
|
||||||
|
|
||||||
|
def _is_dml() -> bool:
|
||||||
|
return any("DmlExecutionProvider" in p for p in modules.globals.execution_providers)
|
||||||
|
|
||||||
|
|
||||||
|
def _analyse_faces(frame: Frame) -> list:
|
||||||
|
"""Run face detection, then recognition (and optionally landmark).
|
||||||
|
|
||||||
|
Replaces InsightFace's ``FaceAnalysis.get()`` to skip the
|
||||||
|
landmark_2d_106 model when only face_swapper is active (saves ~1ms
|
||||||
|
per face and avoids an unnecessary ONNX session call).
|
||||||
|
"""
|
||||||
|
fa = get_face_analyser()
|
||||||
|
|
||||||
|
bboxes, kpss = fa.det_model.detect(frame, max_num=0, metric="default")
|
||||||
|
if bboxes.shape[0] == 0:
|
||||||
|
return []
|
||||||
|
|
||||||
|
need_landmark = _needs_landmark()
|
||||||
|
rec_model = fa.models.get("recognition")
|
||||||
|
lmk_model = fa.models.get("landmark_2d_106") if need_landmark else None
|
||||||
|
|
||||||
|
from insightface.app.common import Face
|
||||||
|
|
||||||
|
faces = []
|
||||||
|
for i in range(bboxes.shape[0]):
|
||||||
|
face = Face(bbox=bboxes[i, 0:4],
|
||||||
|
kps=kpss[i] if kpss is not None else None,
|
||||||
|
det_score=bboxes[i, 4])
|
||||||
|
if rec_model is not None:
|
||||||
|
rec_model.get(frame, face)
|
||||||
|
if lmk_model is not None:
|
||||||
|
lmk_model.get(frame, face)
|
||||||
|
faces.append(face)
|
||||||
|
|
||||||
|
return faces
|
||||||
|
|
||||||
|
|
||||||
|
def get_one_face(frame: Frame, faces: Any = None) -> Any:
|
||||||
|
if faces is None:
|
||||||
|
if _is_dml():
|
||||||
|
with modules.globals.dml_lock:
|
||||||
|
faces = _analyse_faces(frame)
|
||||||
|
else:
|
||||||
|
faces = _analyse_faces(frame)
|
||||||
try:
|
try:
|
||||||
return min(face, key=lambda x: x.bbox[0])
|
return min(faces, key=lambda x: x.bbox[0])
|
||||||
except ValueError:
|
except ValueError:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def get_many_faces(frame: Frame) -> Any:
|
def get_many_faces(frame: Frame) -> Any:
|
||||||
try:
|
try:
|
||||||
return get_face_analyser().get(frame)
|
if _is_dml():
|
||||||
|
with modules.globals.dml_lock:
|
||||||
|
return _analyse_faces(frame)
|
||||||
|
else:
|
||||||
|
return _analyse_faces(frame)
|
||||||
except IndexError:
|
except IndexError:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
def detect_one_face_fast(frame: Frame) -> Any:
|
||||||
|
"""Detection-only — skips landmark and recognition models.
|
||||||
|
|
||||||
|
Returns a Face with bbox, kps, det_score (enough for face swap).
|
||||||
|
~10ms vs ~16ms for full get_one_face() at 1080p.
|
||||||
|
"""
|
||||||
|
from insightface.app.common import Face
|
||||||
|
fa = get_face_analyser()
|
||||||
|
bboxes, kpss = fa.det_model.detect(frame, max_num=0, metric='default')
|
||||||
|
if bboxes.shape[0] == 0:
|
||||||
|
return None
|
||||||
|
idx = int(bboxes[:, 0].argmin())
|
||||||
|
return Face(bbox=bboxes[idx, :4], kps=kpss[idx], det_score=bboxes[idx, 4])
|
||||||
|
|
||||||
|
|
||||||
|
def detect_many_faces_fast(frame: Frame) -> Any:
|
||||||
|
"""Detection-only multi-face — skips landmark and recognition."""
|
||||||
|
from insightface.app.common import Face
|
||||||
|
fa = get_face_analyser()
|
||||||
|
bboxes, kpss = fa.det_model.detect(frame, max_num=0, metric='default')
|
||||||
|
if bboxes.shape[0] == 0:
|
||||||
|
return None
|
||||||
|
return [Face(bbox=bboxes[i, :4], kps=kpss[i], det_score=bboxes[i, 4])
|
||||||
|
for i in range(bboxes.shape[0])]
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_landmarks(frame: Frame, faces: Any) -> None:
|
||||||
|
"""Run the 2d106 landmark model in-place on faces that lack it.
|
||||||
|
|
||||||
|
The fast webcam path (detect_one_face_fast / detect_many_faces_fast)
|
||||||
|
produces detection-only Face objects with no ``landmark_2d_106``.
|
||||||
|
Mouth masking needs those landmarks, so add them on demand only when
|
||||||
|
the feature is active — keeping the fast path fast otherwise.
|
||||||
|
"""
|
||||||
|
if faces is None:
|
||||||
|
return
|
||||||
|
if not isinstance(faces, (list, tuple)):
|
||||||
|
faces = [faces]
|
||||||
|
|
||||||
|
fa = get_face_analyser()
|
||||||
|
lmk_model = fa.models.get("landmark_2d_106")
|
||||||
|
if lmk_model is None:
|
||||||
|
return
|
||||||
|
|
||||||
|
for face in faces:
|
||||||
|
if face is None:
|
||||||
|
continue
|
||||||
|
# insightface Face is a dict; missing keys raise AttributeError,
|
||||||
|
# so getattr(..., None) is the safe presence check.
|
||||||
|
if getattr(face, "landmark_2d_106", None) is None:
|
||||||
|
try:
|
||||||
|
lmk_model.get(frame, face)
|
||||||
|
except Exception as e: # pragma: no cover - never break the swap
|
||||||
|
print(f"Error computing 2d106 landmarks: {e}")
|
||||||
|
|
||||||
|
|
||||||
def has_valid_map() -> bool:
|
def has_valid_map() -> bool:
|
||||||
for map in modules.globals.source_target_map:
|
for map in modules.globals.source_target_map:
|
||||||
if "source" in map and "target" in map:
|
if "source" in map and "target" in map:
|
||||||
@@ -86,8 +257,10 @@ def add_blank_map() -> Any:
|
|||||||
def get_unique_faces_from_target_image() -> Any:
|
def get_unique_faces_from_target_image() -> Any:
|
||||||
try:
|
try:
|
||||||
modules.globals.source_target_map = []
|
modules.globals.source_target_map = []
|
||||||
target_frame = cv2.imread(modules.globals.target_path)
|
target_frame = imread_unicode(modules.globals.target_path)
|
||||||
many_faces = get_many_faces(target_frame)
|
many_faces = get_many_faces(target_frame)
|
||||||
|
if many_faces is None:
|
||||||
|
return None
|
||||||
i = 0
|
i = 0
|
||||||
|
|
||||||
for face in many_faces:
|
for face in many_faces:
|
||||||
@@ -120,8 +293,10 @@ def get_unique_faces_from_target_video() -> Any:
|
|||||||
|
|
||||||
i = 0
|
i = 0
|
||||||
for temp_frame_path in tqdm(temp_frame_paths, desc="Extracting face embeddings from frames"):
|
for temp_frame_path in tqdm(temp_frame_paths, desc="Extracting face embeddings from frames"):
|
||||||
temp_frame = cv2.imread(temp_frame_path)
|
temp_frame = imread_unicode(temp_frame_path)
|
||||||
many_faces = get_many_faces(temp_frame)
|
many_faces = get_many_faces(temp_frame)
|
||||||
|
if many_faces is None:
|
||||||
|
continue
|
||||||
|
|
||||||
for face in many_faces:
|
for face in many_faces:
|
||||||
face_embeddings.append(face.normed_embedding)
|
face_embeddings.append(face.normed_embedding)
|
||||||
@@ -163,6 +338,9 @@ def default_target_face():
|
|||||||
best_frame = frame
|
best_frame = frame
|
||||||
break
|
break
|
||||||
|
|
||||||
|
if best_face is None:
|
||||||
|
continue # No faces detected in this cluster — skip
|
||||||
|
|
||||||
for frame in map['target_faces_in_frame']:
|
for frame in map['target_faces_in_frame']:
|
||||||
for face in frame['faces']:
|
for face in frame['faces']:
|
||||||
if face['det_score'] > best_face['det_score']:
|
if face['det_score'] > best_face['det_score']:
|
||||||
@@ -171,7 +349,7 @@ def default_target_face():
|
|||||||
|
|
||||||
x_min, y_min, x_max, y_max = best_face['bbox']
|
x_min, y_min, x_max, y_max = best_face['bbox']
|
||||||
|
|
||||||
target_frame = cv2.imread(best_frame['location'])
|
target_frame = imread_unicode(best_frame['location'])
|
||||||
map['target'] = {
|
map['target'] = {
|
||||||
'cv2' : target_frame[int(y_min):int(y_max), int(x_min):int(x_max)],
|
'cv2' : target_frame[int(y_min):int(y_max), int(x_min):int(x_max)],
|
||||||
'face' : best_face
|
'face' : best_face
|
||||||
@@ -187,7 +365,7 @@ def dump_faces(centroids: Any, frame_face_embeddings: list):
|
|||||||
Path(temp_directory_path + f"/{i}").mkdir(parents=True, exist_ok=True)
|
Path(temp_directory_path + f"/{i}").mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
for frame in tqdm(frame_face_embeddings, desc=f"Copying faces to temp/./{i}"):
|
for frame in tqdm(frame_face_embeddings, desc=f"Copying faces to temp/./{i}"):
|
||||||
temp_frame = cv2.imread(frame['location'])
|
temp_frame = imread_unicode(frame['location'])
|
||||||
|
|
||||||
j = 0
|
j = 0
|
||||||
for face in frame['faces']:
|
for face in frame['faces']:
|
||||||
@@ -195,5 +373,5 @@ def dump_faces(centroids: Any, frame_face_embeddings: list):
|
|||||||
x_min, y_min, x_max, y_max = face['bbox']
|
x_min, y_min, x_max, y_max = face['bbox']
|
||||||
|
|
||||||
if temp_frame[int(y_min):int(y_max), int(x_min):int(x_max)].size > 0:
|
if temp_frame[int(y_min):int(y_max), int(x_min):int(x_max)].size > 0:
|
||||||
cv2.imwrite(temp_directory_path + f"/{i}/{frame['frame']}_{j}.png", temp_frame[int(y_min):int(y_max), int(x_min):int(x_max)])
|
imwrite_unicode(temp_directory_path + f"/{i}/{frame['frame']}_{j}.png", temp_frame[int(y_min):int(y_max), int(x_min):int(x_max)])
|
||||||
j += 1
|
j += 1
|
||||||
|
|||||||
+11
-4
@@ -6,10 +6,13 @@ from typing import List, Dict, Any
|
|||||||
ROOT_DIR = os.path.dirname(os.path.abspath(__file__))
|
ROOT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||||
WORKFLOW_DIR = os.path.join(ROOT_DIR, "workflow")
|
WORKFLOW_DIR = os.path.join(ROOT_DIR, "workflow")
|
||||||
|
|
||||||
file_types = [
|
# Canonical media extensions, defined once so the file dialogs and
|
||||||
("Image", ("*.png", "*.jpg", "*.jpeg", "*.gif", "*.bmp")),
|
# has_image_extension never drift. GIF is intentionally excluded: OpenCV's
|
||||||
("Video", ("*.mp4", "*.mkv")),
|
# cv2.imread/imwrite (the only image I/O this app uses) cannot decode or
|
||||||
]
|
# encode GIF on 4.10 or 4.11, so offering it would silently fail. WEBP works
|
||||||
|
# via the libwebp bundled with opencv-python.
|
||||||
|
IMAGE_EXTENSIONS = (".png", ".jpg", ".jpeg", ".bmp", ".webp")
|
||||||
|
VIDEO_EXTENSIONS = (".mp4", ".mkv")
|
||||||
|
|
||||||
# Face Mapping Data
|
# Face Mapping Data
|
||||||
source_target_map: List[Dict[str, Any]] = [] # Stores detailed map for image/video processing
|
source_target_map: List[Dict[str, Any]] = [] # Stores detailed map for image/video processing
|
||||||
@@ -63,6 +66,7 @@ show_mouth_mask_box: bool = False # Visualize the mouth mask area (for debuggin
|
|||||||
mask_feather_ratio: int = 12 # Denominator for feathering calculation (higher = smaller feather)
|
mask_feather_ratio: int = 12 # Denominator for feathering calculation (higher = smaller feather)
|
||||||
mask_down_size: float = 0.1 # Expansion factor for lower lip mask (relative)
|
mask_down_size: float = 0.1 # Expansion factor for lower lip mask (relative)
|
||||||
mask_size: float = 1.0 # Expansion factor for upper lip mask (relative)
|
mask_size: float = 1.0 # Expansion factor for upper lip mask (relative)
|
||||||
|
mouth_mask_size: float = 0.0 # Mouth mask size (0-100; 0=off, 100=mouth to chin)
|
||||||
|
|
||||||
# --- START: Added for Frame Interpolation ---
|
# --- START: Added for Frame Interpolation ---
|
||||||
enable_interpolation: bool = True # Toggle temporal smoothing
|
enable_interpolation: bool = True # Toggle temporal smoothing
|
||||||
@@ -70,3 +74,6 @@ interpolation_weight: float = 0 # Blend weight for current frame (0.0-1.0). Low
|
|||||||
# --- END: Added for Frame Interpolation ---
|
# --- END: Added for Frame Interpolation ---
|
||||||
|
|
||||||
# --- END OF FILE globals.py ---
|
# --- END OF FILE globals.py ---
|
||||||
|
|
||||||
|
import threading
|
||||||
|
dml_lock = threading.Lock()
|
||||||
|
|||||||
+21
-22
@@ -18,36 +18,35 @@ Usage
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
import cv2
|
import cv2
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from typing import Tuple, Optional
|
from typing import Tuple
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# CUDA availability detection (evaluated once at import time)
|
# CUDA availability detection (evaluated once at import time)
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
CUDA_AVAILABLE: bool = False
|
CUDA_AVAILABLE: bool = False
|
||||||
|
|
||||||
try:
|
# OpenCV CUDA per-operation acceleration is DISABLED by default.
|
||||||
# cv2.cuda.GpuMat is only present when OpenCV is compiled with CUDA
|
# Each gpu_* call uploads to GPU, processes, then downloads back to CPU.
|
||||||
_test_mat = cv2.cuda.GpuMat()
|
# At webcam resolution (~960x540) this upload/download overhead far exceeds
|
||||||
# Verify we have the required filter / image-processing functions
|
# the time saved on the actual operation, making it slower than pure CPU.
|
||||||
_has_gauss = hasattr(cv2.cuda, "createGaussianFilter")
|
# The heavy lifting (face detection, swap, enhancement) runs on GPU via
|
||||||
_has_resize = hasattr(cv2.cuda, "resize")
|
# ONNX Runtime's CUDAExecutionProvider, which is where GPU matters.
|
||||||
_has_cvt = hasattr(cv2.cuda, "cvtColor")
|
#
|
||||||
if _has_gauss and _has_resize and _has_cvt:
|
# To force-enable, set OPENCV_CUDA_PROCESSING=1 in your environment.
|
||||||
CUDA_AVAILABLE = True
|
if os.environ.get("OPENCV_CUDA_PROCESSING") == "1":
|
||||||
print("[gpu_processing] OpenCV CUDA support detected – GPU-accelerated processing enabled.")
|
try:
|
||||||
else:
|
_test_mat = cv2.cuda.GpuMat()
|
||||||
missing = []
|
_has_gauss = hasattr(cv2.cuda, "createGaussianFilter")
|
||||||
if not _has_gauss:
|
_has_resize = hasattr(cv2.cuda, "resize")
|
||||||
missing.append("createGaussianFilter")
|
_has_cvt = hasattr(cv2.cuda, "cvtColor")
|
||||||
if not _has_resize:
|
if _has_gauss and _has_resize and _has_cvt:
|
||||||
missing.append("resize")
|
CUDA_AVAILABLE = True
|
||||||
if not _has_cvt:
|
print("[gpu_processing] OpenCV CUDA processing enabled via OPENCV_CUDA_PROCESSING=1.")
|
||||||
missing.append("cvtColor")
|
except Exception:
|
||||||
print(f"[gpu_processing] cv2.cuda.GpuMat exists but missing: {', '.join(missing)} – falling back to CPU.")
|
pass
|
||||||
except Exception:
|
|
||||||
print("[gpu_processing] OpenCV CUDA not available – using CPU fallback for all operations.")
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|||||||
+1
-1
@@ -1,3 +1,3 @@
|
|||||||
name = 'Deep-Live-Cam'
|
name = 'Deep-Live-Cam'
|
||||||
version = '2.0.3c'
|
version = '2.1.5'
|
||||||
edition = 'GitHub Edition'
|
edition = 'GitHub Edition'
|
||||||
@@ -0,0 +1,178 @@
|
|||||||
|
import os
|
||||||
|
import platform
|
||||||
|
import ssl
|
||||||
|
import threading
|
||||||
|
import urllib.error
|
||||||
|
import urllib.request
|
||||||
|
from typing import Dict, List, Optional
|
||||||
|
|
||||||
|
from tqdm import tqdm
|
||||||
|
|
||||||
|
from modules.paths import MODELS_DIR
|
||||||
|
|
||||||
|
HF_REPO_ID = "hacksider/deep-live-cam"
|
||||||
|
HF_RESOLVE_BASE = f"https://huggingface.co/{HF_REPO_ID}/resolve/main/"
|
||||||
|
|
||||||
|
MODEL_SIZES: Dict[str, int] = {
|
||||||
|
"inswapper_128.onnx": 554253681,
|
||||||
|
"inswapper_128_fp16.onnx": 277680638,
|
||||||
|
"gfpgan-1024.onnx": 365875079,
|
||||||
|
"GPEN-BFR-256.onnx": 75715262,
|
||||||
|
"GPEN-BFR-512.onnx": 284244491,
|
||||||
|
"buffalo_l/buffalo_l/1k3d68.onnx": 143607619,
|
||||||
|
"buffalo_l/buffalo_l/2d106det.onnx": 5030888,
|
||||||
|
"buffalo_l/buffalo_l/det_10g.onnx": 16923827,
|
||||||
|
"buffalo_l/buffalo_l/genderage.onnx": 1322532,
|
||||||
|
"buffalo_l/buffalo_l/w600k_r50.onnx": 174383860,
|
||||||
|
}
|
||||||
|
|
||||||
|
_LOCKS: Dict[str, threading.Lock] = {}
|
||||||
|
_LOCKS_GUARD = threading.Lock()
|
||||||
|
|
||||||
|
CHUNK_SIZE = 1024 * 256
|
||||||
|
|
||||||
|
|
||||||
|
def _ssl_context():
|
||||||
|
if platform.system().lower() == "darwin":
|
||||||
|
return ssl._create_unverified_context()
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _lock_for(key: str) -> threading.Lock:
|
||||||
|
with _LOCKS_GUARD:
|
||||||
|
if key not in _LOCKS:
|
||||||
|
_LOCKS[key] = threading.Lock()
|
||||||
|
return _LOCKS[key]
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_url(name: str) -> str:
|
||||||
|
return HF_RESOLVE_BASE + name.replace(os.sep, "/")
|
||||||
|
|
||||||
|
|
||||||
|
def local_path(name: str, dest_dir: Optional[str] = None) -> str:
|
||||||
|
if dest_dir is not None:
|
||||||
|
return os.path.join(dest_dir, os.path.basename(name))
|
||||||
|
return os.path.join(MODELS_DIR, *name.replace("/", os.sep).split(os.sep))
|
||||||
|
|
||||||
|
|
||||||
|
def expected_size(name: str) -> Optional[int]:
|
||||||
|
return MODEL_SIZES.get(name.replace(os.sep, "/"))
|
||||||
|
|
||||||
|
|
||||||
|
def is_present(name: str, dest_dir: Optional[str] = None) -> bool:
|
||||||
|
path = local_path(name, dest_dir)
|
||||||
|
return os.path.isfile(path) and os.path.getsize(path) > 0
|
||||||
|
|
||||||
|
|
||||||
|
def _download(name: str, url: str, target: str, size: Optional[int]) -> bool:
|
||||||
|
os.makedirs(os.path.dirname(target) or MODELS_DIR, exist_ok=True)
|
||||||
|
partial = target + ".part"
|
||||||
|
resume_from = os.path.getsize(partial) if os.path.isfile(partial) else 0
|
||||||
|
|
||||||
|
headers = {"User-Agent": "Deep-Live-Cam"}
|
||||||
|
if resume_from:
|
||||||
|
headers["Range"] = f"bytes={resume_from}-"
|
||||||
|
|
||||||
|
try:
|
||||||
|
request = urllib.request.Request(url, headers=headers)
|
||||||
|
response = urllib.request.urlopen(request, context=_ssl_context(), timeout=60)
|
||||||
|
except urllib.error.HTTPError as error:
|
||||||
|
if resume_from and error.code in (416, 501):
|
||||||
|
try:
|
||||||
|
os.remove(partial)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return _download(name, url, target, size)
|
||||||
|
print(f"[DLC.MODELS] Failed to download {name}: HTTP {error.code}")
|
||||||
|
return False
|
||||||
|
except (urllib.error.URLError, OSError) as error:
|
||||||
|
print(f"[DLC.MODELS] Failed to download {name}: {error}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
with response:
|
||||||
|
if resume_from and getattr(response, "status", 200) != 206:
|
||||||
|
resume_from = 0
|
||||||
|
remaining = int(response.headers.get("Content-Length", 0) or 0)
|
||||||
|
total = size or (resume_from + remaining) or None
|
||||||
|
mode = "ab" if resume_from else "wb"
|
||||||
|
try:
|
||||||
|
with open(partial, mode) as handle:
|
||||||
|
with tqdm(
|
||||||
|
total=total,
|
||||||
|
initial=resume_from,
|
||||||
|
desc=f"Downloading {os.path.basename(name)}",
|
||||||
|
unit="B",
|
||||||
|
unit_scale=True,
|
||||||
|
unit_divisor=1024,
|
||||||
|
) as progress:
|
||||||
|
while True:
|
||||||
|
buffer = response.read(CHUNK_SIZE)
|
||||||
|
if not buffer:
|
||||||
|
break
|
||||||
|
handle.write(buffer)
|
||||||
|
progress.update(len(buffer))
|
||||||
|
except (urllib.error.URLError, OSError) as error:
|
||||||
|
print(f"[DLC.MODELS] Download of {name} interrupted: {error}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
downloaded = os.path.getsize(partial)
|
||||||
|
if size is not None and downloaded != size:
|
||||||
|
print(f"[DLC.MODELS] {name} is {downloaded} bytes, expected {size}. Discarding.")
|
||||||
|
try:
|
||||||
|
os.remove(partial)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return False
|
||||||
|
|
||||||
|
try:
|
||||||
|
os.replace(partial, target)
|
||||||
|
except OSError as error:
|
||||||
|
print(f"[DLC.MODELS] Could not finalise {name}: {error}")
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_model(
|
||||||
|
name: str, quiet: bool = False, dest_dir: Optional[str] = None
|
||||||
|
) -> Optional[str]:
|
||||||
|
name = name.replace(os.sep, "/")
|
||||||
|
target = local_path(name, dest_dir)
|
||||||
|
|
||||||
|
with _lock_for(target):
|
||||||
|
if is_present(name, dest_dir):
|
||||||
|
return target
|
||||||
|
if not quiet:
|
||||||
|
print(f"[DLC.MODELS] {name} not found in models folder, downloading...")
|
||||||
|
if _download(name, resolve_url(name), target, expected_size(name)):
|
||||||
|
return target
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_any(names: List[str]) -> Optional[str]:
|
||||||
|
for name in names:
|
||||||
|
if is_present(name):
|
||||||
|
return local_path(name)
|
||||||
|
for name in names:
|
||||||
|
path = ensure_model(name)
|
||||||
|
if path is not None:
|
||||||
|
return path
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_insightface_pack(name: str = "buffalo_l") -> bool:
|
||||||
|
members = [n for n in MODEL_SIZES if n.startswith(f"{name}/")]
|
||||||
|
if not members:
|
||||||
|
return False
|
||||||
|
|
||||||
|
dest_dir = os.path.join(os.path.expanduser("~"), ".insightface", "models", name)
|
||||||
|
if all(is_present(member, dest_dir) for member in members):
|
||||||
|
return True
|
||||||
|
|
||||||
|
print(f"[DLC.MODELS] insightface pack '{name}' is missing, downloading...")
|
||||||
|
ok = True
|
||||||
|
for member in members:
|
||||||
|
if ensure_model(member, quiet=True, dest_dir=dest_dir) is None:
|
||||||
|
ok = False
|
||||||
|
if not ok:
|
||||||
|
print(f"[DLC.MODELS] Could not pre-fill '{name}'; insightface will retry.")
|
||||||
|
return ok
|
||||||
@@ -0,0 +1,550 @@
|
|||||||
|
"""ONNX model optimizations for CoreML execution on Apple Silicon.
|
||||||
|
|
||||||
|
Each pass eliminates a different CPU↔ANE round-trip that ORT's CoreML EP
|
||||||
|
would otherwise introduce:
|
||||||
|
|
||||||
|
1. **Shape/Gather constant folding** — Dynamic ``Shape`` → ``Gather`` chains
|
||||||
|
(e.g. for FPN upsample target sizes in RetinaFace) force ops onto CPU even
|
||||||
|
when the input dimensions are known at load time. We run ONNX shape
|
||||||
|
inference with the known input size and replace these chains with constants.
|
||||||
|
Float32-noise-level differences only (max ~6e-6).
|
||||||
|
|
||||||
|
2. **Pad(reflect) decomposition** — CoreML doesn't support ``Pad(mode=reflect)``.
|
||||||
|
Models using reflect padding (e.g. inswapper_128) get split into many CoreML
|
||||||
|
subgraphs with CPU fallbacks between each. We rewrite each ``Pad(reflect)``
|
||||||
|
as equivalent ``Slice`` + ``Concat`` ops that CoreML handles natively.
|
||||||
|
Bit-for-bit identical output. (Fixed upstream in microsoft/onnxruntime#28073.)
|
||||||
|
|
||||||
|
3. **Split → Slice decomposition** — CoreML's EP doesn't support the ONNX
|
||||||
|
``Split`` op, causing partition boundaries in models with channel-wise
|
||||||
|
splits (e.g. GFPGAN's SFT modulation). Each 2-way Split becomes two Slices.
|
||||||
|
|
||||||
|
4. **Scalar Gather widening** — ORT's CoreML EP rejects ``Gather`` nodes with
|
||||||
|
rank-0 (scalar) indices. StyleGAN-derived models (GFPGAN) slice per-layer
|
||||||
|
style codes using exactly this pattern. We widen each scalar index to
|
||||||
|
``[1]`` and squeeze the added axis on the Gather output.
|
||||||
|
(Filed upstream as microsoft/onnxruntime#28180.)
|
||||||
|
|
||||||
|
All passes are cached on disk with a ``_coreml`` suffix so the rewrite cost
|
||||||
|
is paid only once per model.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import platform
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
IS_APPLE_SILICON = platform.system() == "Darwin" and platform.machine() == "arm64"
|
||||||
|
|
||||||
|
|
||||||
|
def optimize_for_coreml(model_path: str, input_shape: tuple = None) -> str:
|
||||||
|
"""Return path to a CoreML-optimized ONNX model.
|
||||||
|
|
||||||
|
Applies all applicable optimizations and caches the result next to
|
||||||
|
the original model (with ``_coreml`` suffix).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
model_path: Path to the original ONNX model.
|
||||||
|
input_shape: Optional fixed input shape (e.g. ``(1, 3, 640, 640)``).
|
||||||
|
When provided, enables Shape/Gather constant folding.
|
||||||
|
|
||||||
|
Returns the optimized path, or the original path if no optimizations
|
||||||
|
apply or we're not on Apple Silicon.
|
||||||
|
"""
|
||||||
|
if not IS_APPLE_SILICON:
|
||||||
|
return model_path
|
||||||
|
|
||||||
|
base, ext = os.path.splitext(model_path)
|
||||||
|
optimized_path = f"{base}_coreml{ext}"
|
||||||
|
if os.path.exists(optimized_path):
|
||||||
|
if os.path.getmtime(optimized_path) >= os.path.getmtime(model_path):
|
||||||
|
return optimized_path
|
||||||
|
|
||||||
|
import onnx
|
||||||
|
from onnx import numpy_helper
|
||||||
|
|
||||||
|
model = onnx.load(model_path)
|
||||||
|
changed = False
|
||||||
|
|
||||||
|
if _fold_shape_gather(model, input_shape):
|
||||||
|
changed = True
|
||||||
|
|
||||||
|
# TODO(ort>=1.26): drop this pass. Fixed upstream by microsoft/onnxruntime#28073.
|
||||||
|
if _decompose_reflect_pad(model):
|
||||||
|
changed = True
|
||||||
|
|
||||||
|
if _decompose_split(model):
|
||||||
|
changed = True
|
||||||
|
|
||||||
|
# TODO: drop this pass once microsoft/onnxruntime#28180 ships. The CoreML
|
||||||
|
# Gather op builder rejects rank-0 (scalar) indices; we widen them to [1]
|
||||||
|
# + Squeeze so StyleGAN-family models (GFPGAN) stay on ANE.
|
||||||
|
if _rewrite_scalar_gather(model):
|
||||||
|
changed = True
|
||||||
|
|
||||||
|
if not changed:
|
||||||
|
return model_path
|
||||||
|
|
||||||
|
# Preserve insightface's emap convention: the INSwapper class reads
|
||||||
|
# graph.initializer[-1] as the embedding map. If the original model
|
||||||
|
# had a (512, 512) matrix as its last initializer, keep it last.
|
||||||
|
_preserve_emap_position(model, numpy_helper)
|
||||||
|
|
||||||
|
onnx.save(model, optimized_path)
|
||||||
|
return optimized_path
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Pass 1: Fold Shape → Gather chains into constants
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _fold_shape_gather(model, input_shape) -> bool:
|
||||||
|
"""Replace dynamic Shape→Gather chains with constants when input size is known.
|
||||||
|
|
||||||
|
Only removes a Shape node when ALL of its consumers are Gather nodes
|
||||||
|
that are also being folded. This prevents breaking graphs where
|
||||||
|
a Shape output feeds into other ops as well.
|
||||||
|
"""
|
||||||
|
if input_shape is None:
|
||||||
|
return False
|
||||||
|
|
||||||
|
from onnx import numpy_helper, shape_inference
|
||||||
|
|
||||||
|
graph = model.graph
|
||||||
|
|
||||||
|
# Set fixed input dimensions for shape inference
|
||||||
|
inp = graph.input[0]
|
||||||
|
dims = inp.type.tensor_type.shape.dim
|
||||||
|
for i, size in enumerate(input_shape):
|
||||||
|
if i < len(dims):
|
||||||
|
dims[i].dim_value = size
|
||||||
|
|
||||||
|
try:
|
||||||
|
model_inferred = shape_inference.infer_shapes(model)
|
||||||
|
except Exception:
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Extract inferred shapes
|
||||||
|
value_shapes = {}
|
||||||
|
for vi in list(model_inferred.graph.value_info) + list(graph.input) + list(graph.output):
|
||||||
|
shape_dims = vi.type.tensor_type.shape.dim
|
||||||
|
shape = []
|
||||||
|
for d in shape_dims:
|
||||||
|
if d.dim_value > 0:
|
||||||
|
shape.append(d.dim_value)
|
||||||
|
else:
|
||||||
|
shape.append(None)
|
||||||
|
value_shapes[vi.name] = shape
|
||||||
|
|
||||||
|
inits = {init.name: numpy_helper.to_array(init) for init in graph.initializer}
|
||||||
|
|
||||||
|
# Build consumer map: output_name → list of consuming nodes
|
||||||
|
consumers = {}
|
||||||
|
for node in graph.node:
|
||||||
|
for i in node.input:
|
||||||
|
consumers.setdefault(i, []).append(node)
|
||||||
|
|
||||||
|
# Also check graph outputs — an output name consumed by the graph
|
||||||
|
# output list must not be removed
|
||||||
|
graph_output_names = {o.name for o in graph.output}
|
||||||
|
|
||||||
|
# Find Shape nodes with fully-known output
|
||||||
|
shape_constants = {}
|
||||||
|
for node in graph.node:
|
||||||
|
if node.op_type == "Shape":
|
||||||
|
inp_shape = value_shapes.get(node.input[0])
|
||||||
|
if inp_shape and all(isinstance(d, int) for d in inp_shape):
|
||||||
|
shape_constants[node.output[0]] = np.array(inp_shape, dtype=np.int64)
|
||||||
|
|
||||||
|
if not shape_constants:
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Find Gather nodes consuming Shape constants
|
||||||
|
gather_constants = {}
|
||||||
|
for node in graph.node:
|
||||||
|
if node.op_type == "Gather" and node.input[0] in shape_constants:
|
||||||
|
idx_name = node.input[1]
|
||||||
|
if idx_name in inits:
|
||||||
|
idx = int(inits[idx_name])
|
||||||
|
val = int(shape_constants[node.input[0]][idx])
|
||||||
|
gather_constants[node.output[0]] = np.array(val, dtype=np.int64)
|
||||||
|
|
||||||
|
if not gather_constants:
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Determine which Gather nodes to fold (always safe — we replace
|
||||||
|
# the output with a constant initializer)
|
||||||
|
gather_remove_ids = set()
|
||||||
|
for node in graph.node:
|
||||||
|
if node.op_type == "Gather" and node.output[0] in gather_constants:
|
||||||
|
gather_remove_ids.add(id(node))
|
||||||
|
|
||||||
|
# Determine which Shape nodes are safe to remove: only if ALL
|
||||||
|
# consumers of the Shape output are Gather nodes being folded,
|
||||||
|
# and the output isn't a graph output.
|
||||||
|
shape_remove_ids = set()
|
||||||
|
for node in graph.node:
|
||||||
|
if node.op_type == "Shape" and node.output[0] in shape_constants:
|
||||||
|
out_name = node.output[0]
|
||||||
|
if out_name in graph_output_names:
|
||||||
|
continue
|
||||||
|
node_consumers = consumers.get(out_name, [])
|
||||||
|
if all(id(c) in gather_remove_ids for c in node_consumers):
|
||||||
|
shape_remove_ids.add(id(node))
|
||||||
|
|
||||||
|
remove_ids = gather_remove_ids | shape_remove_ids
|
||||||
|
|
||||||
|
# Add Gather output constants as initializers
|
||||||
|
existing = {i.name for i in graph.initializer}
|
||||||
|
for name, val in gather_constants.items():
|
||||||
|
if name not in existing:
|
||||||
|
graph.initializer.append(numpy_helper.from_array(val, name=name))
|
||||||
|
|
||||||
|
new_nodes = [n for n in graph.node if id(n) not in remove_ids]
|
||||||
|
del graph.node[:]
|
||||||
|
graph.node.extend(new_nodes)
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Pass 2: Decompose Pad(reflect) → Slice + Concat
|
||||||
|
#
|
||||||
|
# TEMPORARY: fixed upstream in microsoft/onnxruntime#28073 (merged 2026-04-20).
|
||||||
|
# Once the ORT floor is >= 1.26.0, MLProgram handles Pad(mode=reflect) natively
|
||||||
|
# via MIL tensor_operation.pad and this entire pass can be deleted.
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _decompose_reflect_pad(model) -> bool:
|
||||||
|
"""Rewrite Pad(reflect) as Slice+Concat sequences CoreML can handle."""
|
||||||
|
from onnx import numpy_helper, helper
|
||||||
|
|
||||||
|
graph = model.graph
|
||||||
|
inits = {init.name: numpy_helper.to_array(init) for init in graph.initializer}
|
||||||
|
|
||||||
|
reflect_pads = []
|
||||||
|
for node in graph.node:
|
||||||
|
if node.op_type == "Pad":
|
||||||
|
mode = "constant"
|
||||||
|
for attr in node.attribute:
|
||||||
|
if attr.name == "mode":
|
||||||
|
mode = attr.s.decode()
|
||||||
|
if mode == "reflect" and len(node.input) > 1 and node.input[1] in inits:
|
||||||
|
reflect_pads.append(node)
|
||||||
|
|
||||||
|
if not reflect_pads:
|
||||||
|
return False
|
||||||
|
|
||||||
|
existing_names = {i.name for i in graph.initializer}
|
||||||
|
|
||||||
|
def ensure_const(name, value):
|
||||||
|
if name not in existing_names:
|
||||||
|
graph.initializer.append(
|
||||||
|
numpy_helper.from_array(np.array(value, dtype=np.int64), name=name)
|
||||||
|
)
|
||||||
|
existing_names.add(name)
|
||||||
|
|
||||||
|
ensure_const("_rp_ax2", [2])
|
||||||
|
ensure_const("_rp_ax3", [3])
|
||||||
|
|
||||||
|
max_pad = 0
|
||||||
|
for node in reflect_pads:
|
||||||
|
pads = inits[node.input[1]].tolist()
|
||||||
|
max_pad = max(max_pad, int(pads[2]), int(pads[3]))
|
||||||
|
|
||||||
|
for v in range(1, max_pad + 2):
|
||||||
|
ensure_const(f"_rp_p{v}", [v])
|
||||||
|
ensure_const(f"_rp_n{v}", [-v])
|
||||||
|
|
||||||
|
_counter = [0]
|
||||||
|
|
||||||
|
def uid():
|
||||||
|
_counter[0] += 1
|
||||||
|
return _counter[0]
|
||||||
|
|
||||||
|
pad_ids = {id(n) for n in reflect_pads}
|
||||||
|
pad_init_names = set()
|
||||||
|
|
||||||
|
new_nodes = []
|
||||||
|
for node in graph.node:
|
||||||
|
if id(node) not in pad_ids:
|
||||||
|
new_nodes.append(node)
|
||||||
|
continue
|
||||||
|
|
||||||
|
pads = inits[node.input[1]].tolist()
|
||||||
|
h_pad, w_pad = int(pads[2]), int(pads[3])
|
||||||
|
|
||||||
|
for inp in node.input[1:]:
|
||||||
|
if inp in inits:
|
||||||
|
pad_init_names.add(inp)
|
||||||
|
|
||||||
|
current = node.input[0]
|
||||||
|
|
||||||
|
if h_pad > 0:
|
||||||
|
top = []
|
||||||
|
for i in range(h_pad, 0, -1):
|
||||||
|
name = f"_rp_t{uid()}"
|
||||||
|
new_nodes.append(helper.make_node(
|
||||||
|
"Slice",
|
||||||
|
inputs=[current, f"_rp_p{i}", f"_rp_p{i+1}", "_rp_ax2"],
|
||||||
|
outputs=[name],
|
||||||
|
))
|
||||||
|
top.append(name)
|
||||||
|
|
||||||
|
bot = []
|
||||||
|
for i in range(1, h_pad + 1):
|
||||||
|
name = f"_rp_b{uid()}"
|
||||||
|
new_nodes.append(helper.make_node(
|
||||||
|
"Slice",
|
||||||
|
inputs=[current, f"_rp_n{i+1}", f"_rp_n{i}", "_rp_ax2"],
|
||||||
|
outputs=[name],
|
||||||
|
))
|
||||||
|
bot.append(name)
|
||||||
|
|
||||||
|
h_out = f"_rp_h{uid()}"
|
||||||
|
new_nodes.append(helper.make_node(
|
||||||
|
"Concat", inputs=top + [current] + bot, outputs=[h_out], axis=2
|
||||||
|
))
|
||||||
|
current = h_out
|
||||||
|
|
||||||
|
if w_pad > 0:
|
||||||
|
left = []
|
||||||
|
for i in range(w_pad, 0, -1):
|
||||||
|
name = f"_rp_l{uid()}"
|
||||||
|
new_nodes.append(helper.make_node(
|
||||||
|
"Slice",
|
||||||
|
inputs=[current, f"_rp_p{i}", f"_rp_p{i+1}", "_rp_ax3"],
|
||||||
|
outputs=[name],
|
||||||
|
))
|
||||||
|
left.append(name)
|
||||||
|
|
||||||
|
right = []
|
||||||
|
for i in range(1, w_pad + 1):
|
||||||
|
name = f"_rp_r{uid()}"
|
||||||
|
new_nodes.append(helper.make_node(
|
||||||
|
"Slice",
|
||||||
|
inputs=[current, f"_rp_n{i+1}", f"_rp_n{i}", "_rp_ax3"],
|
||||||
|
outputs=[name],
|
||||||
|
))
|
||||||
|
right.append(name)
|
||||||
|
|
||||||
|
new_nodes.append(helper.make_node(
|
||||||
|
"Concat",
|
||||||
|
inputs=left + [current] + right,
|
||||||
|
outputs=[node.output[0]],
|
||||||
|
axis=3,
|
||||||
|
))
|
||||||
|
elif h_pad > 0:
|
||||||
|
new_nodes.append(helper.make_node(
|
||||||
|
"Identity", inputs=[current], outputs=[node.output[0]]
|
||||||
|
))
|
||||||
|
|
||||||
|
# Remove old Pad initializers
|
||||||
|
clean_inits = [i for i in graph.initializer if i.name not in pad_init_names]
|
||||||
|
del graph.initializer[:]
|
||||||
|
graph.initializer.extend(clean_inits)
|
||||||
|
|
||||||
|
del graph.node[:]
|
||||||
|
graph.node.extend(new_nodes)
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Pass 3: Decompose Split → Slice pairs
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _decompose_split(model) -> bool:
|
||||||
|
"""Rewrite Split(axis=1) as Slice pairs that CoreML can handle.
|
||||||
|
|
||||||
|
CoreML's EP doesn't support the ONNX ``Split`` op, causing partition
|
||||||
|
boundaries in models that use channel-wise splits (e.g. GFPGAN's SFT
|
||||||
|
modulation layers). Each Split with two outputs becomes two Slice ops.
|
||||||
|
"""
|
||||||
|
from onnx import numpy_helper, helper
|
||||||
|
|
||||||
|
graph = model.graph
|
||||||
|
|
||||||
|
splits = []
|
||||||
|
for node in graph.node:
|
||||||
|
if node.op_type == "Split":
|
||||||
|
axis = 0
|
||||||
|
split_sizes = []
|
||||||
|
for attr in node.attribute:
|
||||||
|
if attr.name == "axis":
|
||||||
|
axis = attr.i
|
||||||
|
if attr.name == "split":
|
||||||
|
split_sizes = list(attr.ints)
|
||||||
|
if axis == 1 and len(split_sizes) == 2 and len(node.output) == 2:
|
||||||
|
splits.append((node, split_sizes))
|
||||||
|
|
||||||
|
if not splits:
|
||||||
|
return False
|
||||||
|
|
||||||
|
existing = {i.name for i in graph.initializer}
|
||||||
|
|
||||||
|
def ensure_const(name, value):
|
||||||
|
if name not in existing:
|
||||||
|
graph.initializer.append(
|
||||||
|
numpy_helper.from_array(np.array(value, dtype=np.int64), name=name)
|
||||||
|
)
|
||||||
|
existing.add(name)
|
||||||
|
|
||||||
|
ensure_const("_sp_ax1", [1])
|
||||||
|
|
||||||
|
# Collect all needed boundary constants
|
||||||
|
for _, (a, b) in splits:
|
||||||
|
ensure_const("_sp_s0", [0])
|
||||||
|
ensure_const(f"_sp_s{a}", [a])
|
||||||
|
ensure_const(f"_sp_s{a + b}", [a + b])
|
||||||
|
|
||||||
|
split_ids = {id(node) for node, _ in splits}
|
||||||
|
replacements = {}
|
||||||
|
for node, (a, b) in splits:
|
||||||
|
slice0 = helper.make_node(
|
||||||
|
"Slice",
|
||||||
|
inputs=[node.input[0], "_sp_s0", f"_sp_s{a}", "_sp_ax1"],
|
||||||
|
outputs=[node.output[0]],
|
||||||
|
)
|
||||||
|
slice1 = helper.make_node(
|
||||||
|
"Slice",
|
||||||
|
inputs=[node.input[0], f"_sp_s{a}", f"_sp_s{a + b}", "_sp_ax1"],
|
||||||
|
outputs=[node.output[1]],
|
||||||
|
)
|
||||||
|
replacements[id(node)] = [slice0, slice1]
|
||||||
|
|
||||||
|
new_nodes = []
|
||||||
|
for node in graph.node:
|
||||||
|
if id(node) in split_ids:
|
||||||
|
new_nodes.extend(replacements[id(node)])
|
||||||
|
else:
|
||||||
|
new_nodes.append(node)
|
||||||
|
|
||||||
|
del graph.node[:]
|
||||||
|
graph.node.extend(new_nodes)
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Pass 4: Widen scalar Gather indices to [1] + Squeeze
|
||||||
|
#
|
||||||
|
# TEMPORARY: filed upstream as microsoft/onnxruntime#28180. ORT's CoreML EP
|
||||||
|
# GatherOpBuilder::IsOpSupportedImpl rejects rank-0 (scalar) indices with
|
||||||
|
# `Gather does not support scalar 'indices'`. The builder's own comment
|
||||||
|
# describes the workaround (promote to [1], squeeze the added axis) but
|
||||||
|
# doesn't apply it. We do the same thing at the ONNX level so StyleGAN-
|
||||||
|
# family models (GFPGAN is the hot example — 16 per-layer style-code
|
||||||
|
# slices) don't split the CoreML subgraph. Once the upstream fix ships
|
||||||
|
# and the ORT floor is raised, delete this pass.
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _rewrite_scalar_gather(model) -> bool:
|
||||||
|
"""Rewrite Gather(data, scalar_idx) as Gather(data, [scalar_idx]) + Squeeze.
|
||||||
|
|
||||||
|
Only touches Gather nodes whose index is a rank-0 int64 constant or
|
||||||
|
initializer; everything else passes through unchanged. The rewrite
|
||||||
|
is semantically identical — indices get an added leading axis, the
|
||||||
|
Squeeze removes it after the gather.
|
||||||
|
"""
|
||||||
|
from onnx import numpy_helper, helper, TensorProto
|
||||||
|
|
||||||
|
graph = model.graph
|
||||||
|
|
||||||
|
# Opset 13 moved Squeeze's axes from attribute to input.
|
||||||
|
opset = next(
|
||||||
|
(o.version for o in model.opset_import if o.domain in ("", "ai.onnx")),
|
||||||
|
11,
|
||||||
|
)
|
||||||
|
|
||||||
|
const_values = {}
|
||||||
|
for n in graph.node:
|
||||||
|
if n.op_type == "Constant":
|
||||||
|
for a in n.attribute:
|
||||||
|
if a.name == "value":
|
||||||
|
const_values[n.output[0]] = a.t
|
||||||
|
init_values = {i.name: i for i in graph.initializer}
|
||||||
|
|
||||||
|
def scalar_int64(name):
|
||||||
|
"""Return int value if `name` resolves to a rank-0 int64 constant, else None."""
|
||||||
|
tensor = const_values.get(name) or init_values.get(name)
|
||||||
|
if tensor is None or tensor.data_type != TensorProto.INT64:
|
||||||
|
return None
|
||||||
|
arr = numpy_helper.to_array(tensor)
|
||||||
|
return int(arr) if arr.ndim == 0 else None
|
||||||
|
|
||||||
|
rewrote = 0
|
||||||
|
new_nodes = []
|
||||||
|
for n in graph.node:
|
||||||
|
if n.op_type == "Gather":
|
||||||
|
val = scalar_int64(n.input[1])
|
||||||
|
if val is not None:
|
||||||
|
axis = next((a.i for a in n.attribute if a.name == "axis"), 0)
|
||||||
|
idx_1d_name = f"{n.input[1]}_1d_{rewrote}"
|
||||||
|
idx_const = helper.make_node(
|
||||||
|
"Constant",
|
||||||
|
inputs=[],
|
||||||
|
outputs=[idx_1d_name],
|
||||||
|
value=helper.make_tensor(idx_1d_name, TensorProto.INT64, [1], [val]),
|
||||||
|
)
|
||||||
|
gather_out = f"{n.output[0]}_pre_squeeze_{rewrote}"
|
||||||
|
new_gather = helper.make_node(
|
||||||
|
"Gather",
|
||||||
|
inputs=[n.input[0], idx_1d_name],
|
||||||
|
outputs=[gather_out],
|
||||||
|
name=n.name,
|
||||||
|
axis=axis,
|
||||||
|
)
|
||||||
|
if opset < 13:
|
||||||
|
squeeze = helper.make_node(
|
||||||
|
"Squeeze",
|
||||||
|
inputs=[gather_out],
|
||||||
|
outputs=[n.output[0]],
|
||||||
|
name=(n.name or "gather") + "_squeeze",
|
||||||
|
axes=[axis],
|
||||||
|
)
|
||||||
|
new_nodes.extend([idx_const, new_gather, squeeze])
|
||||||
|
else:
|
||||||
|
axes_name = f"{idx_1d_name}_sq_axes"
|
||||||
|
axes_const = helper.make_node(
|
||||||
|
"Constant",
|
||||||
|
inputs=[],
|
||||||
|
outputs=[axes_name],
|
||||||
|
value=helper.make_tensor(axes_name, TensorProto.INT64, [1], [axis]),
|
||||||
|
)
|
||||||
|
squeeze = helper.make_node(
|
||||||
|
"Squeeze",
|
||||||
|
inputs=[gather_out, axes_name],
|
||||||
|
outputs=[n.output[0]],
|
||||||
|
name=(n.name or "gather") + "_squeeze",
|
||||||
|
)
|
||||||
|
new_nodes.extend([idx_const, axes_const, new_gather, squeeze])
|
||||||
|
rewrote += 1
|
||||||
|
continue
|
||||||
|
new_nodes.append(n)
|
||||||
|
|
||||||
|
if rewrote == 0:
|
||||||
|
return False
|
||||||
|
|
||||||
|
del graph.node[:]
|
||||||
|
graph.node.extend(new_nodes)
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Helpers
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _preserve_emap_position(model, numpy_helper):
|
||||||
|
"""Keep the insightface emap (512×512 matrix) as the last initializer."""
|
||||||
|
graph = model.graph
|
||||||
|
emap_init = None
|
||||||
|
for init in graph.initializer:
|
||||||
|
if not init.name.startswith("_rp_"):
|
||||||
|
arr = numpy_helper.to_array(init)
|
||||||
|
if len(arr.shape) == 2 and arr.shape[0] == 512 and arr.shape[1] == 512:
|
||||||
|
emap_init = init
|
||||||
|
break
|
||||||
|
|
||||||
|
if emap_init is not None:
|
||||||
|
inits = [i for i in graph.initializer if i.name != emap_init.name]
|
||||||
|
del graph.initializer[:]
|
||||||
|
graph.initializer.extend(inits)
|
||||||
|
graph.initializer.append(emap_init)
|
||||||
@@ -0,0 +1,91 @@
|
|||||||
|
"""Centralized platform + accelerator detection.
|
||||||
|
|
||||||
|
Imported once at startup to expose typed flags the rest of the codebase
|
||||||
|
can branch on without re-querying `platform`, `torch.cuda`, or
|
||||||
|
`onnxruntime.get_available_providers()` repeatedly.
|
||||||
|
|
||||||
|
The banner printed by :func:`print_banner` is the single user-facing
|
||||||
|
report of which code path the app will take.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import platform as _platform
|
||||||
|
import sys
|
||||||
|
from typing import List, Tuple
|
||||||
|
|
||||||
|
IS_WINDOWS: bool = _platform.system() == "Windows"
|
||||||
|
IS_MACOS: bool = _platform.system() == "Darwin"
|
||||||
|
IS_LINUX: bool = _platform.system() == "Linux"
|
||||||
|
IS_APPLE_SILICON: bool = IS_MACOS and _platform.machine() == "arm64"
|
||||||
|
|
||||||
|
|
||||||
|
def _detect_torch_cuda() -> bool:
|
||||||
|
try:
|
||||||
|
import torch # noqa: WPS433 — local import, avoid hard dep at module load
|
||||||
|
return bool(torch.cuda.is_available())
|
||||||
|
except Exception:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _detect_onnx_providers() -> List[str]:
|
||||||
|
try:
|
||||||
|
import onnxruntime
|
||||||
|
return list(onnxruntime.get_available_providers())
|
||||||
|
except Exception:
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
HAS_TORCH_CUDA: bool = _detect_torch_cuda()
|
||||||
|
ONNX_PROVIDERS: List[str] = _detect_onnx_providers()
|
||||||
|
HAS_CUDA_PROVIDER: bool = "CUDAExecutionProvider" in ONNX_PROVIDERS
|
||||||
|
HAS_COREML_PROVIDER: bool = "CoreMLExecutionProvider" in ONNX_PROVIDERS
|
||||||
|
HAS_DML_PROVIDER: bool = "DmlExecutionProvider" in ONNX_PROVIDERS
|
||||||
|
HAS_OPENVINO_PROVIDER: bool = "OpenVINOExecutionProvider" in ONNX_PROVIDERS
|
||||||
|
|
||||||
|
# OpenVINO execution-provider config shared by every ONNX session builder.
|
||||||
|
# AUTO:GPU,NPU,CPU lets OpenVINO pick the best available device in priority
|
||||||
|
# order (Intel GPU → NPU → CPU).
|
||||||
|
OPENVINO_PROVIDER_CONFIG = (
|
||||||
|
"OpenVINOExecutionProvider",
|
||||||
|
{"device_type": "AUTO:GPU,NPU,CPU"},
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def camera_backends() -> List[Tuple[int, int]]:
|
||||||
|
"""Return an ordered list of ``(device_index, cv2_backend)`` attempts.
|
||||||
|
|
||||||
|
Windows prefers MSMF (60fps capable) with DirectShow as fallback.
|
||||||
|
macOS/Linux use the default backend (AVFoundation / V4L2).
|
||||||
|
"""
|
||||||
|
import cv2
|
||||||
|
if IS_WINDOWS:
|
||||||
|
return [
|
||||||
|
(0, cv2.CAP_MSMF),
|
||||||
|
(0, cv2.CAP_DSHOW),
|
||||||
|
(0, cv2.CAP_ANY),
|
||||||
|
]
|
||||||
|
return [(0, cv2.CAP_ANY)]
|
||||||
|
|
||||||
|
|
||||||
|
def accelerator_label() -> str:
|
||||||
|
if HAS_CUDA_PROVIDER:
|
||||||
|
return "CUDA (NVIDIA)"
|
||||||
|
if IS_APPLE_SILICON and HAS_COREML_PROVIDER:
|
||||||
|
return "CoreML (Apple Neural Engine)"
|
||||||
|
if HAS_COREML_PROVIDER:
|
||||||
|
return "CoreML"
|
||||||
|
if HAS_OPENVINO_PROVIDER:
|
||||||
|
return "OpenVINO (Intel)"
|
||||||
|
if HAS_DML_PROVIDER:
|
||||||
|
return "DirectML"
|
||||||
|
return "CPU"
|
||||||
|
|
||||||
|
|
||||||
|
def print_banner() -> None:
|
||||||
|
"""Print a one-line summary of the platform + accelerator selection."""
|
||||||
|
os_label = f"{_platform.system()} {_platform.machine()}"
|
||||||
|
print(
|
||||||
|
f"[platform] {os_label} | python {sys.version.split()[0]} | "
|
||||||
|
f"accelerator: {accelerator_label()} | providers: {ONNX_PROVIDERS}",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
@@ -1,4 +1,17 @@
|
|||||||
|
import importlib.util
|
||||||
|
import os
|
||||||
|
|
||||||
import numpy
|
import numpy
|
||||||
|
|
||||||
|
# Keras 3 defaults to the TensorFlow backend, which has no Python 3.14 wheels.
|
||||||
|
# opennsfw2 only runs inference, so any installed backend works; pick one that
|
||||||
|
# is actually present before opennsfw2 imports keras.
|
||||||
|
if "KERAS_BACKEND" not in os.environ:
|
||||||
|
for _backend in ("torch", "tensorflow", "jax"):
|
||||||
|
if importlib.util.find_spec(_backend) is not None:
|
||||||
|
os.environ["KERAS_BACKEND"] = _backend
|
||||||
|
break
|
||||||
|
|
||||||
import opennsfw2
|
import opennsfw2
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
import cv2 # Add OpenCV import
|
import cv2 # Add OpenCV import
|
||||||
|
|||||||
@@ -14,6 +14,7 @@ import numpy as np
|
|||||||
import onnxruntime
|
import onnxruntime
|
||||||
|
|
||||||
import modules.globals
|
import modules.globals
|
||||||
|
from modules.platform_info import OPENVINO_PROVIDER_CONFIG
|
||||||
|
|
||||||
IS_APPLE_SILICON = platform.system() == "Darwin" and platform.machine() == "arm64"
|
IS_APPLE_SILICON = platform.system() == "Darwin" and platform.machine() == "arm64"
|
||||||
|
|
||||||
@@ -21,10 +22,107 @@ IS_APPLE_SILICON = platform.system() == "Darwin" and platform.machine() == "arm6
|
|||||||
THREAD_SEMAPHORE = threading.Semaphore(min(max(1, (os.cpu_count() or 1)), 8))
|
THREAD_SEMAPHORE = threading.Semaphore(min(max(1, (os.cpu_count() or 1)), 8))
|
||||||
|
|
||||||
|
|
||||||
|
def build_provider_config(providers=None):
|
||||||
|
"""Wrap raw provider name strings with optimised CUDA / CoreML options.
|
||||||
|
|
||||||
|
Providers that are already ``(name, options_dict)`` tuples are passed
|
||||||
|
through unchanged. Non-CUDA providers are left as bare strings.
|
||||||
|
"""
|
||||||
|
if providers is None:
|
||||||
|
providers = modules.globals.execution_providers
|
||||||
|
|
||||||
|
config = []
|
||||||
|
for p in providers:
|
||||||
|
if isinstance(p, tuple):
|
||||||
|
# Already configured – pass through
|
||||||
|
config.append(p)
|
||||||
|
elif p == "CUDAExecutionProvider":
|
||||||
|
# Use bare provider — ONNX Runtime's defaults are fastest on
|
||||||
|
# modern GPUs (Blackwell/sm_120). Custom options like
|
||||||
|
# EXHAUSTIVE cudnn_conv_algo_search hurt performance on these
|
||||||
|
# architectures.
|
||||||
|
config.append(p)
|
||||||
|
elif p == "CoreMLExecutionProvider" and IS_APPLE_SILICON:
|
||||||
|
config.append((
|
||||||
|
"CoreMLExecutionProvider",
|
||||||
|
{
|
||||||
|
"ModelFormat": "MLProgram",
|
||||||
|
"MLComputeUnits": "ALL",
|
||||||
|
"AllowLowPrecisionAccumulationOnGPU": 1,
|
||||||
|
},
|
||||||
|
))
|
||||||
|
elif p == "OpenVINOExecutionProvider":
|
||||||
|
# AUTO lets OpenVINO select the best device
|
||||||
|
config.append(OPENVINO_PROVIDER_CONFIG)
|
||||||
|
else:
|
||||||
|
config.append(p)
|
||||||
|
return config
|
||||||
|
|
||||||
|
|
||||||
|
def run_inference(session: onnxruntime.InferenceSession,
|
||||||
|
input_name: str,
|
||||||
|
input_tensor: "np.ndarray") -> "np.ndarray":
|
||||||
|
"""Run ONNX inference, using IO binding when a CUDA session is active.
|
||||||
|
|
||||||
|
IO binding avoids redundant host↔device copies by transferring the
|
||||||
|
input tensor directly to GPU memory and letting ONNX Runtime allocate
|
||||||
|
the output on the device. Falls back to the standard ``session.run``
|
||||||
|
path for non-CUDA providers or if binding fails.
|
||||||
|
"""
|
||||||
|
if "CUDAExecutionProvider" in session.get_providers():
|
||||||
|
try:
|
||||||
|
io_binding = session.io_binding()
|
||||||
|
|
||||||
|
# Input: numpy → GPU
|
||||||
|
ort_input = onnxruntime.OrtValue.ortvalue_from_numpy(
|
||||||
|
input_tensor, "cuda", 0,
|
||||||
|
)
|
||||||
|
io_binding.bind_ortvalue_input(input_name, ort_input)
|
||||||
|
|
||||||
|
# Output: allocate on GPU (avoids a CPU-side allocation)
|
||||||
|
output_name = session.get_outputs()[0].name
|
||||||
|
io_binding.bind_output(output_name, "cuda", 0)
|
||||||
|
|
||||||
|
session.run_with_iobinding(io_binding)
|
||||||
|
|
||||||
|
return io_binding.get_outputs()[0].numpy()
|
||||||
|
except Exception:
|
||||||
|
# Fall back to standard path (e.g. ORT version mismatch,
|
||||||
|
# unsupported op, or VRAM pressure)
|
||||||
|
pass
|
||||||
|
|
||||||
|
return session.run(None, {input_name: input_tensor})[0]
|
||||||
|
|
||||||
|
|
||||||
def create_onnx_session(model_path: str) -> onnxruntime.InferenceSession:
|
def create_onnx_session(model_path: str) -> onnxruntime.InferenceSession:
|
||||||
"""Create an ONNX Runtime session using the configured execution providers."""
|
"""Create an ONNX Runtime session with optimised provider config.
|
||||||
providers = modules.globals.execution_providers
|
|
||||||
session = onnxruntime.InferenceSession(model_path, providers=providers)
|
On Apple Silicon, applies CoreML graph optimizations (Pad decomposition,
|
||||||
|
Shape/Gather folding, Split decomposition) to reduce CPU↔ANE partition
|
||||||
|
boundaries.
|
||||||
|
"""
|
||||||
|
if IS_APPLE_SILICON:
|
||||||
|
from modules.onnx_optimize import optimize_for_coreml
|
||||||
|
# Infer input shape from the model for Shape/Gather folding
|
||||||
|
try:
|
||||||
|
import onnx
|
||||||
|
m = onnx.load(model_path)
|
||||||
|
inp = m.graph.input[0]
|
||||||
|
dims = inp.type.tensor_type.shape.dim
|
||||||
|
shape = tuple(d.dim_value for d in dims if d.dim_value > 0)
|
||||||
|
input_shape = shape if len(shape) == 4 else None
|
||||||
|
except Exception:
|
||||||
|
input_shape = None
|
||||||
|
model_path = optimize_for_coreml(model_path, input_shape=input_shape)
|
||||||
|
|
||||||
|
providers = build_provider_config()
|
||||||
|
session_options = onnxruntime.SessionOptions()
|
||||||
|
session_options.graph_optimization_level = (
|
||||||
|
onnxruntime.GraphOptimizationLevel.ORT_ENABLE_ALL
|
||||||
|
)
|
||||||
|
session = onnxruntime.InferenceSession(
|
||||||
|
model_path, sess_options=session_options, providers=providers,
|
||||||
|
)
|
||||||
return session
|
return session
|
||||||
|
|
||||||
|
|
||||||
@@ -118,7 +216,8 @@ def enhance_face_onnx(
|
|||||||
|
|
||||||
blob = preprocess_face(face_crop, input_size)
|
blob = preprocess_face(face_crop, input_size)
|
||||||
with THREAD_SEMAPHORE:
|
with THREAD_SEMAPHORE:
|
||||||
output = session.run(None, {session.get_inputs()[0].name: blob})[0]
|
input_name = session.get_inputs()[0].name
|
||||||
|
output = run_inference(session, input_name, blob)
|
||||||
enhanced = postprocess_face(output)
|
enhanced = postprocess_face(output)
|
||||||
|
|
||||||
# Create mask for blending (feathered edges)
|
# Create mask for blending (feathered edges)
|
||||||
|
|||||||
@@ -1,12 +1,17 @@
|
|||||||
|
import os
|
||||||
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
import importlib
|
import importlib
|
||||||
from concurrent.futures import ThreadPoolExecutor
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
from types import ModuleType
|
from types import ModuleType
|
||||||
from typing import Any, List, Callable
|
from typing import Any, List, Callable
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
|
|
||||||
import modules
|
import modules
|
||||||
import modules.globals
|
import modules.globals
|
||||||
|
from modules.face_analyser import get_one_face
|
||||||
|
|
||||||
FRAME_PROCESSORS_MODULES: List[ModuleType] = []
|
FRAME_PROCESSORS_MODULES: List[ModuleType] = []
|
||||||
FRAME_PROCESSORS_INTERFACE = [
|
FRAME_PROCESSORS_INTERFACE = [
|
||||||
@@ -32,6 +37,7 @@ def load_frame_processor_module(frame_processor: str) -> Any:
|
|||||||
frame_processor_module = importlib.import_module(f'modules.processors.frame.{frame_processor}')
|
frame_processor_module = importlib.import_module(f'modules.processors.frame.{frame_processor}')
|
||||||
for method_name in FRAME_PROCESSORS_INTERFACE:
|
for method_name in FRAME_PROCESSORS_INTERFACE:
|
||||||
if not hasattr(frame_processor_module, method_name):
|
if not hasattr(frame_processor_module, method_name):
|
||||||
|
print(f"Frame processor {frame_processor} is missing required method {method_name}")
|
||||||
sys.exit()
|
sys.exit()
|
||||||
except ImportError:
|
except ImportError:
|
||||||
print(f"Frame processor {frame_processor} not found")
|
print(f"Frame processor {frame_processor} not found")
|
||||||
@@ -54,7 +60,7 @@ def set_frame_processors_modules_from_ui(frame_processors: List[str]) -> None:
|
|||||||
current_processor_names = [proc.__name__.split('.')[-1] for proc in FRAME_PROCESSORS_MODULES]
|
current_processor_names = [proc.__name__.split('.')[-1] for proc in FRAME_PROCESSORS_MODULES]
|
||||||
|
|
||||||
for frame_processor, state in modules.globals.fp_ui.items():
|
for frame_processor, state in modules.globals.fp_ui.items():
|
||||||
if state == True and frame_processor not in current_processor_names:
|
if state and frame_processor not in current_processor_names:
|
||||||
try:
|
try:
|
||||||
frame_processor_module = load_frame_processor_module(frame_processor)
|
frame_processor_module = load_frame_processor_module(frame_processor)
|
||||||
FRAME_PROCESSORS_MODULES.append(frame_processor_module)
|
FRAME_PROCESSORS_MODULES.append(frame_processor_module)
|
||||||
@@ -65,7 +71,7 @@ def set_frame_processors_modules_from_ui(frame_processors: List[str]) -> None:
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"Warning: Error loading frame processor {frame_processor} requested by UI state: {e}")
|
print(f"Warning: Error loading frame processor {frame_processor} requested by UI state: {e}")
|
||||||
|
|
||||||
elif state == False and frame_processor in current_processor_names:
|
elif not state and frame_processor in current_processor_names:
|
||||||
try:
|
try:
|
||||||
module_to_remove = next((mod for mod in FRAME_PROCESSORS_MODULES if mod.__name__.endswith(f'.{frame_processor}')), None)
|
module_to_remove = next((mod for mod in FRAME_PROCESSORS_MODULES if mod.__name__.endswith(f'.{frame_processor}')), None)
|
||||||
if module_to_remove:
|
if module_to_remove:
|
||||||
@@ -107,3 +113,295 @@ def process_video(source_path: str, frame_paths: list[str], process_frames: Call
|
|||||||
with tqdm(total=total, desc='Processing', unit='frame', dynamic_ncols=True, bar_format=progress_bar_format) as progress:
|
with tqdm(total=total, desc='Processing', unit='frame', dynamic_ncols=True, bar_format=progress_bar_format) as progress:
|
||||||
progress.set_postfix({'execution_providers': modules.globals.execution_providers, 'execution_threads': modules.globals.execution_threads, 'max_memory': modules.globals.max_memory})
|
progress.set_postfix({'execution_providers': modules.globals.execution_providers, 'execution_threads': modules.globals.execution_threads, 'max_memory': modules.globals.max_memory})
|
||||||
multi_process_frame(source_path, frame_paths, process_frames, progress)
|
multi_process_frame(source_path, frame_paths, process_frames, progress)
|
||||||
|
|
||||||
|
|
||||||
|
def process_video_in_memory(source_path: str, target_path: str, fps: float) -> bool:
|
||||||
|
"""Process video frames in-memory using FFmpeg pipes, eliminating disk I/O.
|
||||||
|
|
||||||
|
Reads raw frames from the source video via an FFmpeg decoder pipe, runs each
|
||||||
|
frame through all active frame processors sequentially, and writes the
|
||||||
|
result directly to an FFmpeg encoder pipe. This avoids extracting frames to
|
||||||
|
PNG on disk, which is the biggest I/O bottleneck in the disk-based pipeline.
|
||||||
|
|
||||||
|
Returns True on success, False on failure (caller should fall back to the
|
||||||
|
disk-based pipeline).
|
||||||
|
"""
|
||||||
|
from modules import imread_unicode
|
||||||
|
from modules.face_analyser import get_one_face
|
||||||
|
from modules.utilities import (
|
||||||
|
get_video_dimensions,
|
||||||
|
estimate_frame_count,
|
||||||
|
get_temp_output_path,
|
||||||
|
)
|
||||||
|
|
||||||
|
temp_output_path = get_temp_output_path(target_path)
|
||||||
|
|
||||||
|
# --- Pre-load source face (needed by face_swapper in simple mode) ---
|
||||||
|
source_face = None
|
||||||
|
if source_path and os.path.exists(source_path):
|
||||||
|
source_img = imread_unicode(source_path)
|
||||||
|
if source_img is not None:
|
||||||
|
source_face = get_one_face(source_img)
|
||||||
|
del source_img
|
||||||
|
if source_face is None:
|
||||||
|
print("[DLC.CORE] Warning: No face detected in source image. "
|
||||||
|
"Face swapping will be skipped.")
|
||||||
|
|
||||||
|
# --- Collect frame processors & reset per-video state ---
|
||||||
|
frame_processors = get_frame_processors_modules(modules.globals.frame_processors)
|
||||||
|
for fp in frame_processors:
|
||||||
|
if hasattr(fp, 'PREVIOUS_FRAME_RESULT'):
|
||||||
|
fp.PREVIOUS_FRAME_RESULT = None
|
||||||
|
|
||||||
|
# --- Video metadata ---
|
||||||
|
try:
|
||||||
|
width, height = get_video_dimensions(target_path)
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[DLC.CORE] Failed to get video dimensions: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
total_frames = estimate_frame_count(target_path, fps)
|
||||||
|
frame_size = width * height * 3
|
||||||
|
|
||||||
|
# --- Build encoder arguments ---
|
||||||
|
encoder = modules.globals.video_encoder
|
||||||
|
encoder_options: List[str] = []
|
||||||
|
is_hw_encoder = False
|
||||||
|
|
||||||
|
if 'CUDAExecutionProvider' in modules.globals.execution_providers:
|
||||||
|
if encoder == 'libx264':
|
||||||
|
encoder = 'h264_nvenc'
|
||||||
|
is_hw_encoder = True
|
||||||
|
encoder_options = [
|
||||||
|
'-preset', 'p4', '-tune', 'hq', '-rc', 'vbr',
|
||||||
|
'-cq', str(modules.globals.video_quality), '-b:v', '0',
|
||||||
|
]
|
||||||
|
elif encoder == 'libx265':
|
||||||
|
encoder = 'hevc_nvenc'
|
||||||
|
is_hw_encoder = True
|
||||||
|
encoder_options = [
|
||||||
|
'-preset', 'p4', '-tune', 'hq', '-rc', 'vbr',
|
||||||
|
'-cq', str(modules.globals.video_quality), '-b:v', '0',
|
||||||
|
]
|
||||||
|
elif 'DmlExecutionProvider' in modules.globals.execution_providers:
|
||||||
|
if encoder == 'libx264':
|
||||||
|
encoder = 'h264_amf'
|
||||||
|
is_hw_encoder = True
|
||||||
|
encoder_options = [
|
||||||
|
'-quality', 'quality', '-rc', 'vbr_latency',
|
||||||
|
'-qp_i', str(modules.globals.video_quality),
|
||||||
|
'-qp_p', str(modules.globals.video_quality),
|
||||||
|
]
|
||||||
|
elif encoder == 'libx265':
|
||||||
|
encoder = 'hevc_amf'
|
||||||
|
is_hw_encoder = True
|
||||||
|
encoder_options = [
|
||||||
|
'-quality', 'quality', '-rc', 'vbr_latency',
|
||||||
|
'-qp_i', str(modules.globals.video_quality),
|
||||||
|
'-qp_p', str(modules.globals.video_quality),
|
||||||
|
]
|
||||||
|
|
||||||
|
if not is_hw_encoder:
|
||||||
|
if encoder == 'libx264':
|
||||||
|
encoder_options = [
|
||||||
|
'-preset', 'medium',
|
||||||
|
'-crf', str(modules.globals.video_quality),
|
||||||
|
'-tune', 'film',
|
||||||
|
]
|
||||||
|
elif encoder == 'libx265':
|
||||||
|
encoder_options = [
|
||||||
|
'-preset', 'medium',
|
||||||
|
'-crf', str(modules.globals.video_quality),
|
||||||
|
'-x265-params', 'log-level=error',
|
||||||
|
]
|
||||||
|
elif encoder == 'libvpx-vp9':
|
||||||
|
encoder_options = [
|
||||||
|
'-crf', str(modules.globals.video_quality),
|
||||||
|
'-b:v', '0', '-cpu-used', '2',
|
||||||
|
]
|
||||||
|
|
||||||
|
# --- Attempt pipeline (hw encoder first, then sw fallback) ---
|
||||||
|
encoders_to_try = [(encoder, encoder_options)]
|
||||||
|
if is_hw_encoder:
|
||||||
|
# Software fallback
|
||||||
|
sw_encoder = 'libx264'
|
||||||
|
sw_options = [
|
||||||
|
'-preset', 'medium',
|
||||||
|
'-crf', str(modules.globals.video_quality),
|
||||||
|
'-tune', 'film',
|
||||||
|
]
|
||||||
|
encoders_to_try.append((sw_encoder, sw_options))
|
||||||
|
|
||||||
|
for attempt, (enc, enc_opts) in enumerate(encoders_to_try):
|
||||||
|
# Reset interpolation state on retry
|
||||||
|
if attempt > 0:
|
||||||
|
for fp in frame_processors:
|
||||||
|
if hasattr(fp, 'PREVIOUS_FRAME_RESULT'):
|
||||||
|
fp.PREVIOUS_FRAME_RESULT = None
|
||||||
|
|
||||||
|
success = _run_pipe_pipeline(
|
||||||
|
target_path, temp_output_path, fps,
|
||||||
|
source_face, frame_processors,
|
||||||
|
width, height, frame_size, total_frames,
|
||||||
|
enc, enc_opts,
|
||||||
|
)
|
||||||
|
if success:
|
||||||
|
return True
|
||||||
|
|
||||||
|
if attempt == 0 and is_hw_encoder:
|
||||||
|
print(f"[DLC.CORE] Hardware encoder '{enc}' failed, "
|
||||||
|
f"retrying with software encoder...")
|
||||||
|
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _run_pipe_pipeline(
|
||||||
|
target_path: str,
|
||||||
|
temp_output_path: str,
|
||||||
|
fps: float,
|
||||||
|
source_face: Any,
|
||||||
|
frame_processors: List[Any],
|
||||||
|
width: int,
|
||||||
|
height: int,
|
||||||
|
frame_size: int,
|
||||||
|
total_frames: int,
|
||||||
|
encoder: str,
|
||||||
|
encoder_options: List[str],
|
||||||
|
) -> bool:
|
||||||
|
"""Run the FFmpeg-pipe read → process → encode pipeline once."""
|
||||||
|
|
||||||
|
# --- Reader: decode source video to raw BGR24 on stdout ---
|
||||||
|
reader_cmd = [
|
||||||
|
'ffmpeg', '-hide_banner',
|
||||||
|
'-hwaccel', 'auto',
|
||||||
|
'-i', target_path,
|
||||||
|
'-f', 'rawvideo',
|
||||||
|
'-pix_fmt', 'bgr24',
|
||||||
|
'-v', 'error',
|
||||||
|
'-',
|
||||||
|
]
|
||||||
|
|
||||||
|
# --- Writer: encode raw BGR24 from stdin ---
|
||||||
|
writer_cmd = [
|
||||||
|
'ffmpeg', '-hide_banner',
|
||||||
|
'-f', 'rawvideo',
|
||||||
|
'-pix_fmt', 'bgr24',
|
||||||
|
'-s', f'{width}x{height}',
|
||||||
|
'-r', str(fps),
|
||||||
|
'-i', '-',
|
||||||
|
'-c:v', encoder,
|
||||||
|
]
|
||||||
|
writer_cmd.extend(encoder_options)
|
||||||
|
writer_cmd.extend([
|
||||||
|
'-pix_fmt', 'yuv420p',
|
||||||
|
'-movflags', '+faststart',
|
||||||
|
'-vf', 'colorspace=bt709:iall=bt601-6-625:fast=1',
|
||||||
|
'-v', 'error',
|
||||||
|
'-y', temp_output_path,
|
||||||
|
])
|
||||||
|
|
||||||
|
reader = None
|
||||||
|
writer = None
|
||||||
|
try:
|
||||||
|
reader = subprocess.Popen(
|
||||||
|
reader_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||||
|
)
|
||||||
|
writer = subprocess.Popen(
|
||||||
|
writer_cmd, stdin=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[DLC.CORE] Failed to start FFmpeg pipes: {e}")
|
||||||
|
for proc in (reader, writer):
|
||||||
|
if proc:
|
||||||
|
try:
|
||||||
|
proc.kill()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return False
|
||||||
|
|
||||||
|
processed_count = 0
|
||||||
|
bar_fmt = ('{l_bar}{bar}| {n_fmt}/{total_fmt} '
|
||||||
|
'[{elapsed}<{remaining}, {rate_fmt}{postfix}]')
|
||||||
|
|
||||||
|
try:
|
||||||
|
with tqdm(total=total_frames, desc='Processing', unit='frame',
|
||||||
|
dynamic_ncols=True, bar_format=bar_fmt) as progress:
|
||||||
|
progress.set_postfix({
|
||||||
|
'execution_providers': modules.globals.execution_providers,
|
||||||
|
'threads': modules.globals.execution_threads,
|
||||||
|
'mode': 'in-memory',
|
||||||
|
})
|
||||||
|
|
||||||
|
# Pipelined detection: while processing frame N (swap on
|
||||||
|
# ANE), start detecting the face in the next frame
|
||||||
|
# (detection on GPU). They use different hardware units
|
||||||
|
# so the work overlaps.
|
||||||
|
detect_executor = ThreadPoolExecutor(max_workers=1)
|
||||||
|
pending_detect = None
|
||||||
|
use_pipeline = not modules.globals.many_faces
|
||||||
|
|
||||||
|
while True:
|
||||||
|
raw = reader.stdout.read(frame_size)
|
||||||
|
if len(raw) != frame_size:
|
||||||
|
break
|
||||||
|
|
||||||
|
frame = np.frombuffer(raw, dtype=np.uint8).reshape(
|
||||||
|
(height, width, 3)
|
||||||
|
).copy()
|
||||||
|
|
||||||
|
# Get the detection result for THIS frame
|
||||||
|
if use_pipeline:
|
||||||
|
if pending_detect is not None:
|
||||||
|
target_face = pending_detect.result()
|
||||||
|
else:
|
||||||
|
target_face = get_one_face(frame)
|
||||||
|
# Start detecting on THIS frame eagerly — the result
|
||||||
|
# will be used for the next iteration. At video
|
||||||
|
# frame rates the face barely moves between frames.
|
||||||
|
# Hand the detector its own copy: the frame processors
|
||||||
|
# below mutate `frame` in place (paste-back), which
|
||||||
|
# would otherwise race with detection.
|
||||||
|
pending_detect = detect_executor.submit(
|
||||||
|
get_one_face, frame.copy())
|
||||||
|
else:
|
||||||
|
target_face = None
|
||||||
|
|
||||||
|
# Run frame through every active processor
|
||||||
|
for fp in frame_processors:
|
||||||
|
try:
|
||||||
|
frame = fp.process_frame(source_face, frame, target_face=target_face)
|
||||||
|
except TypeError:
|
||||||
|
frame = fp.process_frame(source_face, frame)
|
||||||
|
|
||||||
|
writer.stdin.write(frame.tobytes())
|
||||||
|
processed_count += 1
|
||||||
|
progress.update(1)
|
||||||
|
|
||||||
|
detect_executor.shutdown(wait=True)
|
||||||
|
|
||||||
|
# Graceful shutdown
|
||||||
|
writer.stdin.close()
|
||||||
|
writer.wait()
|
||||||
|
reader.wait()
|
||||||
|
|
||||||
|
if writer.returncode != 0:
|
||||||
|
stderr_out = writer.stderr.read().decode(errors='ignore').strip()
|
||||||
|
if stderr_out:
|
||||||
|
print(f"[DLC.CORE] FFmpeg encoder error: {stderr_out}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
return processed_count > 0 and os.path.isfile(temp_output_path)
|
||||||
|
|
||||||
|
except BrokenPipeError:
|
||||||
|
print("[DLC.CORE] FFmpeg pipe broken (encoder may not be available).")
|
||||||
|
return False
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[DLC.CORE] In-memory processing error: {e}")
|
||||||
|
return False
|
||||||
|
finally:
|
||||||
|
for proc in (reader, writer):
|
||||||
|
if proc:
|
||||||
|
try:
|
||||||
|
proc.kill()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|||||||
@@ -1,4 +1,3 @@
|
|||||||
# --- START OF FILE face_enhancer.py ---
|
|
||||||
# Uses ONNX Runtime for GFPGAN face enhancement (no torch/gfpgan dependency)
|
# Uses ONNX Runtime for GFPGAN face enhancement (no torch/gfpgan dependency)
|
||||||
|
|
||||||
from typing import Any, List
|
from typing import Any, List
|
||||||
@@ -11,8 +10,9 @@ import onnxruntime
|
|||||||
|
|
||||||
import modules.globals
|
import modules.globals
|
||||||
import modules.processors.frame.core
|
import modules.processors.frame.core
|
||||||
|
from modules import imread_unicode, imwrite_unicode
|
||||||
from modules.core import update_status
|
from modules.core import update_status
|
||||||
from modules.face_analyser import get_one_face, get_many_faces
|
from modules.face_analyser import get_many_faces
|
||||||
from modules.typing import Frame, Face
|
from modules.typing import Frame, Face
|
||||||
from modules.utilities import (
|
from modules.utilities import (
|
||||||
is_image,
|
is_image,
|
||||||
@@ -23,6 +23,7 @@ FACE_ENHANCER = None
|
|||||||
THREAD_SEMAPHORE = threading.Semaphore()
|
THREAD_SEMAPHORE = threading.Semaphore()
|
||||||
THREAD_LOCK = threading.Lock()
|
THREAD_LOCK = threading.Lock()
|
||||||
NAME = "DLC.FACE-ENHANCER"
|
NAME = "DLC.FACE-ENHANCER"
|
||||||
|
MODEL_FILE = "gfpgan-1024.onnx"
|
||||||
|
|
||||||
abs_dir = os.path.dirname(os.path.abspath(__file__))
|
abs_dir = os.path.dirname(os.path.abspath(__file__))
|
||||||
models_dir = os.path.join(
|
models_dir = os.path.join(
|
||||||
@@ -44,11 +45,12 @@ FFHQ_TEMPLATE_512 = np.array(
|
|||||||
|
|
||||||
|
|
||||||
def pre_check() -> bool:
|
def pre_check() -> bool:
|
||||||
model_path = os.path.join(models_dir, "gfpgan-1024.onnx")
|
from modules.model_downloader import ensure_model
|
||||||
if not os.path.exists(model_path):
|
|
||||||
|
if ensure_model(MODEL_FILE) is None:
|
||||||
update_status(
|
update_status(
|
||||||
f"GFPGAN ONNX model not found at {model_path}. "
|
f"Could not obtain {MODEL_FILE}. Place it in the models folder "
|
||||||
"Please place gfpgan-1024.onnx in the models folder.",
|
"manually or check your internet connection.",
|
||||||
NAME,
|
NAME,
|
||||||
)
|
)
|
||||||
return False
|
return False
|
||||||
@@ -73,26 +75,23 @@ def get_face_enhancer() -> onnxruntime.InferenceSession:
|
|||||||
|
|
||||||
with THREAD_LOCK:
|
with THREAD_LOCK:
|
||||||
if FACE_ENHANCER is None:
|
if FACE_ENHANCER is None:
|
||||||
model_path = os.path.join(models_dir, "gfpgan-1024.onnx")
|
from modules.model_downloader import ensure_model
|
||||||
|
|
||||||
if not os.path.exists(model_path):
|
model_path = ensure_model(MODEL_FILE)
|
||||||
|
|
||||||
|
if model_path is None:
|
||||||
raise FileNotFoundError(
|
raise FileNotFoundError(
|
||||||
f"{NAME}: Model not found at {model_path}"
|
f"{NAME}: Model not found at "
|
||||||
|
f"{os.path.join(models_dir, MODEL_FILE)} and could not be "
|
||||||
|
"downloaded"
|
||||||
)
|
)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
providers = modules.globals.execution_providers
|
from modules.processors.frame._onnx_enhancer import (
|
||||||
|
create_onnx_session,
|
||||||
session_options = onnxruntime.SessionOptions()
|
|
||||||
session_options.graph_optimization_level = (
|
|
||||||
onnxruntime.GraphOptimizationLevel.ORT_ENABLE_ALL
|
|
||||||
)
|
)
|
||||||
|
|
||||||
FACE_ENHANCER = onnxruntime.InferenceSession(
|
FACE_ENHANCER = create_onnx_session(model_path)
|
||||||
model_path,
|
|
||||||
sess_options=session_options,
|
|
||||||
providers=providers,
|
|
||||||
)
|
|
||||||
|
|
||||||
input_info = FACE_ENHANCER.get_inputs()[0]
|
input_info = FACE_ENHANCER.get_inputs()[0]
|
||||||
output_info = FACE_ENHANCER.get_outputs()[0]
|
output_info = FACE_ENHANCER.get_outputs()[0]
|
||||||
@@ -158,6 +157,18 @@ def _align_face(
|
|||||||
return aligned_face, affine_matrix
|
return aligned_face, affine_matrix
|
||||||
|
|
||||||
|
|
||||||
|
_HAS_TORCH_CUDA = False
|
||||||
|
try:
|
||||||
|
import torch
|
||||||
|
if torch.cuda.is_available():
|
||||||
|
_HAS_TORCH_CUDA = True
|
||||||
|
except ImportError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Cache the feathered mask — it's the same for every call at a given size
|
||||||
|
_enhancer_cache: dict = {'mask': None, 'mask_size': 0}
|
||||||
|
|
||||||
|
|
||||||
def _paste_back(
|
def _paste_back(
|
||||||
frame: Frame,
|
frame: Frame,
|
||||||
enhanced_face: np.ndarray,
|
enhanced_face: np.ndarray,
|
||||||
@@ -167,53 +178,77 @@ def _paste_back(
|
|||||||
"""
|
"""
|
||||||
Paste an enhanced (aligned) face back onto the original frame using the
|
Paste an enhanced (aligned) face back onto the original frame using the
|
||||||
inverse affine transform with feathered-edge blending.
|
inverse affine transform with feathered-edge blending.
|
||||||
|
|
||||||
|
Optimized: operates on a tight crop around the face bbox instead of the
|
||||||
|
full frame, and uses GPU for blending when available.
|
||||||
"""
|
"""
|
||||||
h, w = frame.shape[:2]
|
h, w = frame.shape[:2]
|
||||||
|
|
||||||
# Inverse the affine warp
|
|
||||||
inv_matrix = cv2.invertAffineTransform(affine_matrix)
|
inv_matrix = cv2.invertAffineTransform(affine_matrix)
|
||||||
inv_restored = cv2.warpAffine(
|
|
||||||
enhanced_face,
|
# Build or reuse cached feathered mask (uint8 — blended via cv2 SIMD ops)
|
||||||
inv_matrix,
|
if _enhancer_cache['mask_size'] != output_size:
|
||||||
(w, h),
|
face_mask_f = np.ones((output_size, output_size), dtype=np.float32)
|
||||||
borderMode=cv2.BORDER_CONSTANT,
|
border = max(1, int(output_size * 0.05))
|
||||||
borderValue=(0, 0, 0),
|
ramp_up = np.linspace(0.0, 1.0, border, dtype=np.float32)
|
||||||
|
ramp_down = np.linspace(1.0, 0.0, border, dtype=np.float32)
|
||||||
|
face_mask_f[:border, :] *= ramp_up[:, None]
|
||||||
|
face_mask_f[-border:, :] *= ramp_down[:, None]
|
||||||
|
face_mask_f[:, :border] *= ramp_up[None, :]
|
||||||
|
face_mask_f[:, -border:] *= ramp_down[None, :]
|
||||||
|
_enhancer_cache['mask'] = (face_mask_f * 255.0).astype(np.uint8)
|
||||||
|
_enhancer_cache['mask_size'] = output_size
|
||||||
|
|
||||||
|
# Compute tight bbox from affine corners (avoids full-frame warpAffine scan)
|
||||||
|
corners = np.array([[0, 0], [output_size, 0],
|
||||||
|
[output_size, output_size], [0, output_size]],
|
||||||
|
dtype=np.float32)
|
||||||
|
transformed = (inv_matrix[:, :2] @ corners.T).T + inv_matrix[:, 2]
|
||||||
|
x1 = max(0, int(np.floor(transformed[:, 0].min())))
|
||||||
|
x2 = min(w, int(np.ceil(transformed[:, 0].max())))
|
||||||
|
y1 = max(0, int(np.floor(transformed[:, 1].min())))
|
||||||
|
y2 = min(h, int(np.ceil(transformed[:, 1].max())))
|
||||||
|
if x1 >= x2 or y1 >= y2:
|
||||||
|
return frame
|
||||||
|
|
||||||
|
# Pad a few pixels for feathering
|
||||||
|
pad = max(1, int(output_size * 0.05)) + 2
|
||||||
|
y1p, y2p = max(0, y1 - pad), min(h, y2 + pad)
|
||||||
|
x1p, x2p = max(0, x1 - pad), min(w, x2 + pad)
|
||||||
|
crop_w, crop_h = x2p - x1p, y2p - y1p
|
||||||
|
|
||||||
|
# Warp enhanced face and mask into crop space only
|
||||||
|
inv_crop = inv_matrix.copy()
|
||||||
|
inv_crop[0, 2] -= x1p
|
||||||
|
inv_crop[1, 2] -= y1p
|
||||||
|
|
||||||
|
inv_restored_crop = cv2.warpAffine(
|
||||||
|
enhanced_face, inv_crop, (crop_w, crop_h),
|
||||||
|
borderMode=cv2.BORDER_CONSTANT, borderValue=(0, 0, 0),
|
||||||
|
)
|
||||||
|
inv_mask_crop = cv2.warpAffine(
|
||||||
|
_enhancer_cache['mask'], inv_crop, (crop_w, crop_h),
|
||||||
|
borderMode=cv2.BORDER_CONSTANT, borderValue=0,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Build a soft feathered mask in aligned space for edge blending
|
target_crop = frame[y1p:y2p, x1p:x2p]
|
||||||
face_mask = np.ones((output_size, output_size), dtype=np.float32)
|
|
||||||
|
|
||||||
# Feather the border (5 % of the size on each edge)
|
if _HAS_TORCH_CUDA:
|
||||||
border = max(1, int(output_size * 0.05))
|
# Upload uint8 alpha — smaller transfer, scale on device.
|
||||||
ramp_up = np.linspace(0.0, 1.0, border, dtype=np.float32)
|
mask_t = torch.from_numpy(inv_mask_crop).cuda().float().mul_(1.0 / 255.0).unsqueeze(2)
|
||||||
ramp_down = np.linspace(1.0, 0.0, border, dtype=np.float32)
|
enhanced_t = torch.from_numpy(inv_restored_crop).float().cuda()
|
||||||
|
target_t = torch.from_numpy(target_crop).float().cuda()
|
||||||
|
blended = (mask_t * enhanced_t + (1.0 - mask_t) * target_t
|
||||||
|
).to(torch.uint8).cpu().numpy()
|
||||||
|
frame[y1p:y2p, x1p:x2p] = blended
|
||||||
|
else:
|
||||||
|
# Fused uint8 blend via cv2 SIMD — ~7× faster than the float32 round-trip.
|
||||||
|
alpha_3c = cv2.merge([inv_mask_crop, inv_mask_crop, inv_mask_crop])
|
||||||
|
inv_alpha = 255 - alpha_3c
|
||||||
|
a_enh = cv2.multiply(inv_restored_crop, alpha_3c, scale=1.0 / 255.0)
|
||||||
|
a_tgt = cv2.multiply(target_crop, inv_alpha, scale=1.0 / 255.0)
|
||||||
|
frame[y1p:y2p, x1p:x2p] = cv2.add(a_enh, a_tgt)
|
||||||
|
|
||||||
# Top / bottom rows
|
return frame
|
||||||
face_mask[:border, :] *= ramp_up[:, None]
|
|
||||||
face_mask[-border:, :] *= ramp_down[:, None]
|
|
||||||
# Left / right columns
|
|
||||||
face_mask[:, :border] *= ramp_up[None, :]
|
|
||||||
face_mask[:, -border:] *= ramp_down[None, :]
|
|
||||||
|
|
||||||
# Expand to 3-channel
|
|
||||||
face_mask_3c = np.stack([face_mask] * 3, axis=-1)
|
|
||||||
|
|
||||||
# Warp mask back to original frame space
|
|
||||||
inv_mask = cv2.warpAffine(
|
|
||||||
face_mask_3c,
|
|
||||||
inv_matrix,
|
|
||||||
(w, h),
|
|
||||||
borderMode=cv2.BORDER_CONSTANT,
|
|
||||||
borderValue=(0, 0, 0),
|
|
||||||
)
|
|
||||||
inv_mask = np.clip(inv_mask, 0.0, 1.0)
|
|
||||||
|
|
||||||
# Alpha-blend
|
|
||||||
result = (
|
|
||||||
frame.astype(np.float32) * (1.0 - inv_mask)
|
|
||||||
+ inv_restored.astype(np.float32) * inv_mask
|
|
||||||
)
|
|
||||||
return np.clip(result, 0, 255).astype(np.uint8)
|
|
||||||
|
|
||||||
|
|
||||||
def _preprocess_face(aligned_face: np.ndarray) -> np.ndarray:
|
def _preprocess_face(aligned_face: np.ndarray) -> np.ndarray:
|
||||||
@@ -221,14 +256,13 @@ def _preprocess_face(aligned_face: np.ndarray) -> np.ndarray:
|
|||||||
Convert an aligned BGR uint8 face image to the ONNX model input tensor.
|
Convert an aligned BGR uint8 face image to the ONNX model input tensor.
|
||||||
Format: NCHW float32, normalised to [-1, 1].
|
Format: NCHW float32, normalised to [-1, 1].
|
||||||
"""
|
"""
|
||||||
# BGR -> RGB
|
# BGR -> RGB, normalize, and transpose in one pass
|
||||||
rgb = cv2.cvtColor(aligned_face, cv2.COLOR_BGR2RGB).astype(np.float32)
|
# Fused: (x / 255.0 - 0.5) / 0.5 = x / 127.5 - 1.0
|
||||||
# [0, 255] -> [0, 1] -> [-1, 1]
|
rgb = aligned_face[:, :, ::-1] # BGR->RGB zero-copy view
|
||||||
rgb = rgb / 255.0
|
chw = np.transpose(rgb, (2, 0, 1)).astype(np.float32)
|
||||||
rgb = (rgb - 0.5) / 0.5
|
chw *= (1.0 / 127.5)
|
||||||
# HWC -> CHW, add batch dim
|
chw -= 1.0
|
||||||
chw = np.transpose(rgb, (2, 0, 1))
|
return chw[np.newaxis, ...] # shape: (1, 3, H, W)
|
||||||
return np.expand_dims(chw, axis=0) # shape: (1, 3, H, W)
|
|
||||||
|
|
||||||
|
|
||||||
def _postprocess_face(output: np.ndarray) -> np.ndarray:
|
def _postprocess_face(output: np.ndarray) -> np.ndarray:
|
||||||
@@ -236,24 +270,42 @@ def _postprocess_face(output: np.ndarray) -> np.ndarray:
|
|||||||
Convert the ONNX model output tensor back to a BGR uint8 image.
|
Convert the ONNX model output tensor back to a BGR uint8 image.
|
||||||
Expects input in NCHW format with values in [-1, 1].
|
Expects input in NCHW format with values in [-1, 1].
|
||||||
"""
|
"""
|
||||||
face = np.squeeze(output) # remove batch dim -> (3, H, W)
|
# Fused: ((x + 1.0) / 2.0) * 255 = (x + 1.0) * 127.5
|
||||||
face = np.transpose(face, (1, 2, 0)) # CHW -> HWC
|
face = output[0] # remove batch dim -> (3, H, W)
|
||||||
# [-1, 1] -> [0, 1] -> [0, 255]
|
face = (face + 1.0) * 127.5
|
||||||
face = (face + 1.0) / 2.0
|
np.clip(face, 0, 255, out=face)
|
||||||
face = np.clip(face * 255.0, 0, 255).astype(np.uint8)
|
face = face.astype(np.uint8).transpose(1, 2, 0) # CHW -> HWC
|
||||||
# RGB -> BGR
|
return face[:, :, ::-1].copy() # RGB -> BGR
|
||||||
return cv2.cvtColor(face, cv2.COLOR_RGB2BGR)
|
|
||||||
|
|
||||||
|
|
||||||
def enhance_face(temp_frame: Frame) -> Frame:
|
# Cache for temporal enhancement skipping in live mode.
|
||||||
"""Enhances all faces in a frame using the GFPGAN ONNX model."""
|
# GFPGAN output barely changes between consecutive frames (same face,
|
||||||
|
# same position), so we run inference every _ENH_INTERVAL frames and
|
||||||
|
# reuse the cached enhanced face + affine matrix in between.
|
||||||
|
_enh_live_cache: dict = {
|
||||||
|
'enhanced_bgr': None,
|
||||||
|
'affine_matrix': None,
|
||||||
|
'align_size': 0,
|
||||||
|
'frame_count': 0,
|
||||||
|
}
|
||||||
|
_ENH_INTERVAL = 2 # run inference every N frames, paste cached result otherwise
|
||||||
|
|
||||||
|
|
||||||
|
def enhance_face(temp_frame: Frame, detected_faces=None) -> Frame:
|
||||||
|
"""Enhances all faces in a frame using the GFPGAN ONNX model.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
detected_faces: Pre-detected face list. When provided, skips
|
||||||
|
the internal detection call (saves ~15-20ms per frame).
|
||||||
|
Also enables temporal caching — inference runs every
|
||||||
|
_ENH_INTERVAL frames, reusing the cached result otherwise.
|
||||||
|
"""
|
||||||
session = get_face_enhancer()
|
session = get_face_enhancer()
|
||||||
|
|
||||||
# Determine model input resolution from the session metadata
|
# Determine model input resolution from the session metadata
|
||||||
input_info = session.get_inputs()[0]
|
input_info = session.get_inputs()[0]
|
||||||
input_name = input_info.name
|
input_name = input_info.name
|
||||||
input_shape = input_info.shape # e.g. [1, 3, 512, 512]
|
input_shape = input_info.shape # e.g. [1, 3, 512, 512]
|
||||||
# Safely extract input size (handle dynamic / symbolic dimensions)
|
|
||||||
try:
|
try:
|
||||||
align_size = int(input_shape[2])
|
align_size = int(input_shape[2])
|
||||||
if align_size <= 0:
|
if align_size <= 0:
|
||||||
@@ -261,15 +313,25 @@ def enhance_face(temp_frame: Frame) -> Frame:
|
|||||||
except (ValueError, TypeError, IndexError):
|
except (ValueError, TypeError, IndexError):
|
||||||
align_size = 512
|
align_size = 512
|
||||||
|
|
||||||
# Detect faces using InsightFace (already a project dependency)
|
# Use pre-detected faces if available, otherwise detect
|
||||||
faces = get_many_faces(temp_frame)
|
faces = detected_faces if detected_faces is not None else get_many_faces(temp_frame)
|
||||||
if not faces:
|
if not faces:
|
||||||
return temp_frame
|
return temp_frame
|
||||||
|
|
||||||
result_frame = temp_frame.copy()
|
# Temporal caching: only available when faces are pre-detected (live mode)
|
||||||
|
# AND we're in single-face mode — the cache holds exactly one enhancement,
|
||||||
|
# so reusing it in many_faces mode would paste the same face onto every
|
||||||
|
# detected target.
|
||||||
|
many_faces_mode = getattr(modules.globals, "many_faces", False)
|
||||||
|
use_cache = detected_faces is not None and not many_faces_mode
|
||||||
|
if use_cache:
|
||||||
|
_enh_live_cache['frame_count'] += 1
|
||||||
|
run_inference_this_frame = (_enh_live_cache['frame_count'] % _ENH_INTERVAL == 0
|
||||||
|
or _enh_live_cache['enhanced_bgr'] is None)
|
||||||
|
else:
|
||||||
|
run_inference_this_frame = True
|
||||||
|
|
||||||
for face in faces:
|
for face in faces:
|
||||||
# Need the 5-point key-points for alignment
|
|
||||||
if not hasattr(face, "kps") or face.kps is None:
|
if not hasattr(face, "kps") or face.kps is None:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
@@ -277,48 +339,68 @@ def enhance_face(temp_frame: Frame) -> Frame:
|
|||||||
if landmarks_5.shape[0] < 5:
|
if landmarks_5.shape[0] < 5:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# Align / crop the face at the model's INPUT resolution
|
if run_inference_this_frame:
|
||||||
aligned_face, affine_matrix = _align_face(
|
aligned_face, affine_matrix = _align_face(
|
||||||
temp_frame, landmarks_5, output_size=align_size
|
temp_frame, landmarks_5, output_size=align_size
|
||||||
)
|
|
||||||
if aligned_face is None or affine_matrix is None:
|
|
||||||
continue
|
|
||||||
|
|
||||||
try:
|
|
||||||
with THREAD_SEMAPHORE:
|
|
||||||
input_tensor = _preprocess_face(aligned_face)
|
|
||||||
output_tensor = session.run(None, {input_name: input_tensor})[0]
|
|
||||||
enhanced_bgr = _postprocess_face(output_tensor)
|
|
||||||
|
|
||||||
# The model may output at a different resolution than its input
|
|
||||||
# (e.g. input 512x512 → output 1024x1024). Resize the enhanced
|
|
||||||
# face back to the alignment size so the inverse affine maps
|
|
||||||
# correctly.
|
|
||||||
eh, ew = enhanced_bgr.shape[:2]
|
|
||||||
if eh != align_size or ew != align_size:
|
|
||||||
enhanced_bgr = cv2.resize(
|
|
||||||
enhanced_bgr,
|
|
||||||
(align_size, align_size),
|
|
||||||
interpolation=cv2.INTER_LANCZOS4,
|
|
||||||
)
|
|
||||||
|
|
||||||
# Paste enhanced face back onto the frame
|
|
||||||
result_frame = _paste_back(
|
|
||||||
result_frame, enhanced_bgr, affine_matrix, output_size=align_size
|
|
||||||
)
|
)
|
||||||
except Exception as e:
|
if aligned_face is None or affine_matrix is None:
|
||||||
print(f"{NAME}: Error enhancing a face: {e}")
|
continue
|
||||||
continue
|
|
||||||
|
|
||||||
return result_frame
|
try:
|
||||||
|
with THREAD_SEMAPHORE:
|
||||||
|
from modules.processors.frame._onnx_enhancer import (
|
||||||
|
run_inference,
|
||||||
|
)
|
||||||
|
input_tensor = _preprocess_face(aligned_face)
|
||||||
|
output_tensor = run_inference(session, input_name, input_tensor)
|
||||||
|
enhanced_bgr = _postprocess_face(output_tensor)
|
||||||
|
|
||||||
|
eh, ew = enhanced_bgr.shape[:2]
|
||||||
|
if eh != align_size or ew != align_size:
|
||||||
|
enhanced_bgr = cv2.resize(
|
||||||
|
enhanced_bgr,
|
||||||
|
(align_size, align_size),
|
||||||
|
interpolation=cv2.INTER_LANCZOS4,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Cache for reuse on next frame
|
||||||
|
if use_cache:
|
||||||
|
_enh_live_cache['enhanced_bgr'] = enhanced_bgr
|
||||||
|
_enh_live_cache['affine_matrix'] = affine_matrix
|
||||||
|
_enh_live_cache['align_size'] = align_size
|
||||||
|
|
||||||
|
_paste_back(
|
||||||
|
temp_frame, enhanced_bgr, affine_matrix, output_size=align_size
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
print(f"{NAME}: Error enhancing a face: {e}")
|
||||||
|
continue
|
||||||
|
else:
|
||||||
|
# Reuse cached enhanced face — just paste back onto current frame
|
||||||
|
cached = _enh_live_cache
|
||||||
|
if cached['enhanced_bgr'] is not None:
|
||||||
|
_paste_back(
|
||||||
|
temp_frame, cached['enhanced_bgr'],
|
||||||
|
cached['affine_matrix'],
|
||||||
|
output_size=cached['align_size'],
|
||||||
|
)
|
||||||
|
if not many_faces_mode:
|
||||||
|
break # single-face live mode — only process first face
|
||||||
|
|
||||||
def process_frame(source_face: Face | None, temp_frame: Frame) -> Frame:
|
|
||||||
"""Processes a frame: enhances face if detected."""
|
|
||||||
temp_frame = enhance_face(temp_frame)
|
|
||||||
return temp_frame
|
return temp_frame
|
||||||
|
|
||||||
|
|
||||||
|
def process_frame(source_face: Face | None, temp_frame: Frame,
|
||||||
|
detected_faces=None) -> Frame:
|
||||||
|
"""Processes a frame: enhances face if detected."""
|
||||||
|
return enhance_face(temp_frame, detected_faces=detected_faces)
|
||||||
|
|
||||||
|
|
||||||
|
def process_frame_v2(temp_frame: Frame, detected_faces=None) -> Frame:
|
||||||
|
"""Processes a frame without source face (used by live webcam preview)."""
|
||||||
|
return enhance_face(temp_frame, detected_faces=detected_faces)
|
||||||
|
|
||||||
|
|
||||||
def process_frames(
|
def process_frames(
|
||||||
source_path: str | None, temp_frame_paths: List[str], progress: Any = None
|
source_path: str | None, temp_frame_paths: List[str], progress: Any = None
|
||||||
) -> None:
|
) -> None:
|
||||||
@@ -332,7 +414,7 @@ def process_frames(
|
|||||||
progress.update(1)
|
progress.update(1)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
temp_frame = cv2.imread(temp_frame_path)
|
temp_frame = imread_unicode(temp_frame_path)
|
||||||
if temp_frame is None:
|
if temp_frame is None:
|
||||||
print(
|
print(
|
||||||
f"{NAME}: Warning: Failed to read frame {temp_frame_path}, skipping."
|
f"{NAME}: Warning: Failed to read frame {temp_frame_path}, skipping."
|
||||||
@@ -342,7 +424,7 @@ def process_frames(
|
|||||||
continue
|
continue
|
||||||
|
|
||||||
result_frame = process_frame(None, temp_frame)
|
result_frame = process_frame(None, temp_frame)
|
||||||
cv2.imwrite(temp_frame_path, result_frame)
|
imwrite_unicode(temp_frame_path, result_frame)
|
||||||
if progress:
|
if progress:
|
||||||
progress.update(1)
|
progress.update(1)
|
||||||
|
|
||||||
@@ -351,12 +433,12 @@ def process_image(
|
|||||||
source_path: str | None, target_path: str, output_path: str
|
source_path: str | None, target_path: str, output_path: str
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Processes a single image file."""
|
"""Processes a single image file."""
|
||||||
target_frame = cv2.imread(target_path)
|
target_frame = imread_unicode(target_path)
|
||||||
if target_frame is None:
|
if target_frame is None:
|
||||||
print(f"{NAME}: Error: Failed to read target image {target_path}")
|
print(f"{NAME}: Error: Failed to read target image {target_path}")
|
||||||
return
|
return
|
||||||
result_frame = process_frame(None, target_frame)
|
result_frame = process_frame(None, target_frame)
|
||||||
cv2.imwrite(output_path, result_frame)
|
imwrite_unicode(output_path, result_frame)
|
||||||
print(f"{NAME}: Enhanced image saved to {output_path}")
|
print(f"{NAME}: Enhanced image saved to {output_path}")
|
||||||
|
|
||||||
|
|
||||||
@@ -367,6 +449,3 @@ def process_video(
|
|||||||
modules.processors.frame.core.process_video(
|
modules.processors.frame.core.process_video(
|
||||||
source_path, temp_frame_paths, process_frames
|
source_path, temp_frame_paths, process_frames
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
# --- END OF FILE face_enhancer.py ---
|
|
||||||
|
|||||||
@@ -4,11 +4,9 @@ from typing import Any, List
|
|||||||
import os
|
import os
|
||||||
import threading
|
import threading
|
||||||
|
|
||||||
import cv2
|
|
||||||
import numpy as np
|
|
||||||
|
|
||||||
import modules.globals
|
import modules.globals
|
||||||
import modules.processors.frame.core
|
import modules.processors.frame.core
|
||||||
|
from modules import imread_unicode, imwrite_unicode
|
||||||
from modules.core import update_status
|
from modules.core import update_status
|
||||||
from modules.face_analyser import get_one_face
|
from modules.face_analyser import get_one_face
|
||||||
from modules.typing import Frame, Face
|
from modules.typing import Frame, Face
|
||||||
@@ -24,7 +22,7 @@ from modules.processors.frame._onnx_enhancer import (
|
|||||||
|
|
||||||
NAME = "DLC.FACE-ENHANCER-GPEN256"
|
NAME = "DLC.FACE-ENHANCER-GPEN256"
|
||||||
INPUT_SIZE = 256
|
INPUT_SIZE = 256
|
||||||
MODEL_URL = "https://github.com/harisreedhar/Face-Upscalers-ONNX/releases/download/GPEN-BFR/GPEN-BFR-256.onnx"
|
MODEL_MIRROR_URL = "https://github.com/harisreedhar/Face-Upscalers-ONNX/releases/download/GPEN-BFR/GPEN-BFR-256.onnx"
|
||||||
MODEL_FILE = "GPEN-BFR-256.onnx"
|
MODEL_FILE = "GPEN-BFR-256.onnx"
|
||||||
|
|
||||||
ENHANCER = None
|
ENHANCER = None
|
||||||
@@ -36,12 +34,33 @@ models_dir = os.path.join(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _obtain_model():
|
||||||
|
from modules.model_downloader import ensure_model
|
||||||
|
|
||||||
|
model_path = ensure_model(MODEL_FILE)
|
||||||
|
if model_path is not None:
|
||||||
|
return model_path
|
||||||
|
|
||||||
|
update_status(f"Retrying {MODEL_FILE} from the mirror...", NAME)
|
||||||
|
from modules.utilities import conditional_download
|
||||||
|
|
||||||
|
try:
|
||||||
|
conditional_download(models_dir, [MODEL_MIRROR_URL])
|
||||||
|
except Exception as error:
|
||||||
|
update_status(f"Mirror download failed: {error}", NAME)
|
||||||
|
return None
|
||||||
|
fallback = os.path.join(models_dir, MODEL_FILE)
|
||||||
|
return fallback if os.path.exists(fallback) else None
|
||||||
|
|
||||||
|
|
||||||
def pre_check() -> bool:
|
def pre_check() -> bool:
|
||||||
model_path = os.path.join(models_dir, MODEL_FILE)
|
if _obtain_model() is None:
|
||||||
if not os.path.exists(model_path):
|
update_status(
|
||||||
update_status(f"Downloading {MODEL_FILE}...", NAME)
|
f"Could not obtain {MODEL_FILE}. Place it in the models folder "
|
||||||
from modules.utilities import conditional_download
|
"manually or check your internet connection.",
|
||||||
conditional_download(models_dir, [MODEL_URL])
|
NAME,
|
||||||
|
)
|
||||||
|
return False
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
@@ -56,12 +75,11 @@ def get_enhancer() -> Any:
|
|||||||
global ENHANCER
|
global ENHANCER
|
||||||
with THREAD_LOCK:
|
with THREAD_LOCK:
|
||||||
if ENHANCER is None:
|
if ENHANCER is None:
|
||||||
model_path = os.path.join(models_dir, MODEL_FILE)
|
model_path = _obtain_model()
|
||||||
if not os.path.exists(model_path):
|
if model_path is None:
|
||||||
from modules.utilities import conditional_download
|
raise FileNotFoundError(
|
||||||
conditional_download(models_dir, [MODEL_URL])
|
f"Model file not found: {os.path.join(models_dir, MODEL_FILE)}"
|
||||||
if not os.path.exists(model_path):
|
)
|
||||||
raise FileNotFoundError(f"Model file not found: {model_path}")
|
|
||||||
print(f"{NAME}: Loading ONNX model from {model_path}")
|
print(f"{NAME}: Loading ONNX model from {model_path}")
|
||||||
ENHANCER = create_onnx_session(model_path)
|
ENHANCER = create_onnx_session(model_path)
|
||||||
warmup_session(ENHANCER)
|
warmup_session(ENHANCER)
|
||||||
@@ -82,8 +100,11 @@ def enhance_face(temp_frame: Frame, face: Face) -> Frame:
|
|||||||
return temp_frame
|
return temp_frame
|
||||||
|
|
||||||
|
|
||||||
def process_frame(source_face: Face | None, temp_frame: Frame) -> Frame:
|
def process_frame(source_face: Face | None, temp_frame: Frame, detected_faces=None) -> Frame:
|
||||||
target_face = get_one_face(temp_frame)
|
if detected_faces:
|
||||||
|
target_face = detected_faces[0]
|
||||||
|
else:
|
||||||
|
target_face = get_one_face(temp_frame)
|
||||||
if target_face is None:
|
if target_face is None:
|
||||||
return temp_frame
|
return temp_frame
|
||||||
return enhance_face(temp_frame, target_face)
|
return enhance_face(temp_frame, target_face)
|
||||||
@@ -100,24 +121,24 @@ def process_frames(
|
|||||||
source_path: str | None, temp_frame_paths: List[str], progress: Any = None
|
source_path: str | None, temp_frame_paths: List[str], progress: Any = None
|
||||||
) -> None:
|
) -> None:
|
||||||
for temp_frame_path in temp_frame_paths:
|
for temp_frame_path in temp_frame_paths:
|
||||||
temp_frame = cv2.imread(temp_frame_path)
|
temp_frame = imread_unicode(temp_frame_path)
|
||||||
if temp_frame is None:
|
if temp_frame is None:
|
||||||
if progress:
|
if progress:
|
||||||
progress.update(1)
|
progress.update(1)
|
||||||
continue
|
continue
|
||||||
result = process_frame(None, temp_frame)
|
result = process_frame(None, temp_frame)
|
||||||
cv2.imwrite(temp_frame_path, result)
|
imwrite_unicode(temp_frame_path, result)
|
||||||
if progress:
|
if progress:
|
||||||
progress.update(1)
|
progress.update(1)
|
||||||
|
|
||||||
|
|
||||||
def process_image(source_path: str | None, target_path: str, output_path: str) -> None:
|
def process_image(source_path: str | None, target_path: str, output_path: str) -> None:
|
||||||
target_frame = cv2.imread(target_path)
|
target_frame = imread_unicode(target_path)
|
||||||
if target_frame is None:
|
if target_frame is None:
|
||||||
print(f"{NAME}: Error: Failed to read target image {target_path}")
|
print(f"{NAME}: Error: Failed to read target image {target_path}")
|
||||||
return
|
return
|
||||||
result_frame = process_frame(None, target_frame)
|
result_frame = process_frame(None, target_frame)
|
||||||
cv2.imwrite(output_path, result_frame)
|
imwrite_unicode(output_path, result_frame)
|
||||||
print(f"{NAME}: Enhanced image saved to {output_path}")
|
print(f"{NAME}: Enhanced image saved to {output_path}")
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -4,11 +4,9 @@ from typing import Any, List
|
|||||||
import os
|
import os
|
||||||
import threading
|
import threading
|
||||||
|
|
||||||
import cv2
|
|
||||||
import numpy as np
|
|
||||||
|
|
||||||
import modules.globals
|
import modules.globals
|
||||||
import modules.processors.frame.core
|
import modules.processors.frame.core
|
||||||
|
from modules import imread_unicode, imwrite_unicode
|
||||||
from modules.core import update_status
|
from modules.core import update_status
|
||||||
from modules.face_analyser import get_one_face
|
from modules.face_analyser import get_one_face
|
||||||
from modules.typing import Frame, Face
|
from modules.typing import Frame, Face
|
||||||
@@ -24,7 +22,7 @@ from modules.processors.frame._onnx_enhancer import (
|
|||||||
|
|
||||||
NAME = "DLC.FACE-ENHANCER-GPEN512"
|
NAME = "DLC.FACE-ENHANCER-GPEN512"
|
||||||
INPUT_SIZE = 512
|
INPUT_SIZE = 512
|
||||||
MODEL_URL = "https://github.com/harisreedhar/Face-Upscalers-ONNX/releases/download/GPEN-BFR/GPEN-BFR-512.onnx"
|
MODEL_MIRROR_URL = "https://github.com/harisreedhar/Face-Upscalers-ONNX/releases/download/GPEN-BFR/GPEN-BFR-512.onnx"
|
||||||
MODEL_FILE = "GPEN-BFR-512.onnx"
|
MODEL_FILE = "GPEN-BFR-512.onnx"
|
||||||
|
|
||||||
ENHANCER = None
|
ENHANCER = None
|
||||||
@@ -36,12 +34,33 @@ models_dir = os.path.join(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _obtain_model():
|
||||||
|
from modules.model_downloader import ensure_model
|
||||||
|
|
||||||
|
model_path = ensure_model(MODEL_FILE)
|
||||||
|
if model_path is not None:
|
||||||
|
return model_path
|
||||||
|
|
||||||
|
update_status(f"Retrying {MODEL_FILE} from the mirror...", NAME)
|
||||||
|
from modules.utilities import conditional_download
|
||||||
|
|
||||||
|
try:
|
||||||
|
conditional_download(models_dir, [MODEL_MIRROR_URL])
|
||||||
|
except Exception as error:
|
||||||
|
update_status(f"Mirror download failed: {error}", NAME)
|
||||||
|
return None
|
||||||
|
fallback = os.path.join(models_dir, MODEL_FILE)
|
||||||
|
return fallback if os.path.exists(fallback) else None
|
||||||
|
|
||||||
|
|
||||||
def pre_check() -> bool:
|
def pre_check() -> bool:
|
||||||
model_path = os.path.join(models_dir, MODEL_FILE)
|
if _obtain_model() is None:
|
||||||
if not os.path.exists(model_path):
|
update_status(
|
||||||
update_status(f"Downloading {MODEL_FILE}...", NAME)
|
f"Could not obtain {MODEL_FILE}. Place it in the models folder "
|
||||||
from modules.utilities import conditional_download
|
"manually or check your internet connection.",
|
||||||
conditional_download(models_dir, [MODEL_URL])
|
NAME,
|
||||||
|
)
|
||||||
|
return False
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
@@ -56,12 +75,11 @@ def get_enhancer() -> Any:
|
|||||||
global ENHANCER
|
global ENHANCER
|
||||||
with THREAD_LOCK:
|
with THREAD_LOCK:
|
||||||
if ENHANCER is None:
|
if ENHANCER is None:
|
||||||
model_path = os.path.join(models_dir, MODEL_FILE)
|
model_path = _obtain_model()
|
||||||
if not os.path.exists(model_path):
|
if model_path is None:
|
||||||
from modules.utilities import conditional_download
|
raise FileNotFoundError(
|
||||||
conditional_download(models_dir, [MODEL_URL])
|
f"Model file not found: {os.path.join(models_dir, MODEL_FILE)}"
|
||||||
if not os.path.exists(model_path):
|
)
|
||||||
raise FileNotFoundError(f"Model file not found: {model_path}")
|
|
||||||
print(f"{NAME}: Loading ONNX model from {model_path}")
|
print(f"{NAME}: Loading ONNX model from {model_path}")
|
||||||
ENHANCER = create_onnx_session(model_path)
|
ENHANCER = create_onnx_session(model_path)
|
||||||
warmup_session(ENHANCER)
|
warmup_session(ENHANCER)
|
||||||
@@ -82,8 +100,11 @@ def enhance_face(temp_frame: Frame, face: Face) -> Frame:
|
|||||||
return temp_frame
|
return temp_frame
|
||||||
|
|
||||||
|
|
||||||
def process_frame(source_face: Face | None, temp_frame: Frame) -> Frame:
|
def process_frame(source_face: Face | None, temp_frame: Frame, detected_faces=None) -> Frame:
|
||||||
target_face = get_one_face(temp_frame)
|
if detected_faces:
|
||||||
|
target_face = detected_faces[0]
|
||||||
|
else:
|
||||||
|
target_face = get_one_face(temp_frame)
|
||||||
if target_face is None:
|
if target_face is None:
|
||||||
return temp_frame
|
return temp_frame
|
||||||
return enhance_face(temp_frame, target_face)
|
return enhance_face(temp_frame, target_face)
|
||||||
@@ -100,24 +121,24 @@ def process_frames(
|
|||||||
source_path: str | None, temp_frame_paths: List[str], progress: Any = None
|
source_path: str | None, temp_frame_paths: List[str], progress: Any = None
|
||||||
) -> None:
|
) -> None:
|
||||||
for temp_frame_path in temp_frame_paths:
|
for temp_frame_path in temp_frame_paths:
|
||||||
temp_frame = cv2.imread(temp_frame_path)
|
temp_frame = imread_unicode(temp_frame_path)
|
||||||
if temp_frame is None:
|
if temp_frame is None:
|
||||||
if progress:
|
if progress:
|
||||||
progress.update(1)
|
progress.update(1)
|
||||||
continue
|
continue
|
||||||
result = process_frame(None, temp_frame)
|
result = process_frame(None, temp_frame)
|
||||||
cv2.imwrite(temp_frame_path, result)
|
imwrite_unicode(temp_frame_path, result)
|
||||||
if progress:
|
if progress:
|
||||||
progress.update(1)
|
progress.update(1)
|
||||||
|
|
||||||
|
|
||||||
def process_image(source_path: str | None, target_path: str, output_path: str) -> None:
|
def process_image(source_path: str | None, target_path: str, output_path: str) -> None:
|
||||||
target_frame = cv2.imread(target_path)
|
target_frame = imread_unicode(target_path)
|
||||||
if target_frame is None:
|
if target_frame is None:
|
||||||
print(f"{NAME}: Error: Failed to read target image {target_path}")
|
print(f"{NAME}: Error: Failed to read target image {target_path}")
|
||||||
return
|
return
|
||||||
result_frame = process_frame(None, target_frame)
|
result_frame = process_frame(None, target_frame)
|
||||||
cv2.imwrite(output_path, result_frame)
|
imwrite_unicode(output_path, result_frame)
|
||||||
print(f"{NAME}: Enhanced image saved to {output_path}")
|
print(f"{NAME}: Enhanced image saved to {output_path}")
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ import cv2
|
|||||||
import numpy as np
|
import numpy as np
|
||||||
from modules.typing import Face, Frame
|
from modules.typing import Face, Frame
|
||||||
import modules.globals
|
import modules.globals
|
||||||
from modules.gpu_processing import gpu_gaussian_blur, gpu_resize, gpu_cvt_color
|
from modules.gpu_processing import gpu_gaussian_blur, gpu_resize
|
||||||
|
|
||||||
def apply_color_transfer(source, target):
|
def apply_color_transfer(source, target):
|
||||||
"""
|
"""
|
||||||
@@ -82,8 +82,8 @@ def create_lower_mouth_mask(
|
|||||||
|
|
||||||
landmarks = face.landmark_2d_106
|
landmarks = face.landmark_2d_106
|
||||||
if landmarks is not None:
|
if landmarks is not None:
|
||||||
# Use outer mouth landmarks (52-63) to capture the lips only
|
# Use outer mouth landmarks (52-71) to capture the full mouth area
|
||||||
lower_lip_order = list(range(52, 64))
|
lower_lip_order = list(range(52, 72))
|
||||||
|
|
||||||
if max(lower_lip_order) >= landmarks.shape[0]:
|
if max(lower_lip_order) >= landmarks.shape[0]:
|
||||||
return mask, mouth_cutout, mouth_box, lower_lip_polygon
|
return mask, mouth_cutout, mouth_box, lower_lip_polygon
|
||||||
@@ -94,13 +94,16 @@ def create_lower_mouth_mask(
|
|||||||
center = np.mean(lower_lip_landmarks, axis=0)
|
center = np.mean(lower_lip_landmarks, axis=0)
|
||||||
|
|
||||||
# Expand the landmarks outward using the mouth_mask_size
|
# Expand the landmarks outward using the mouth_mask_size
|
||||||
# Use a more conservative expansion to avoid affecting face shape
|
mouth_mask_size = getattr(modules.globals, "mouth_mask_size", 0.0) # 0-100 slider
|
||||||
expansion_factor = (
|
expansion_factor = 1 + (mouth_mask_size / 100.0) * 2.5
|
||||||
1 + modules.globals.mask_down_size * modules.globals.mouth_mask_size
|
|
||||||
)
|
|
||||||
expanded_landmarks = (lower_lip_landmarks - center) * expansion_factor + center
|
|
||||||
|
|
||||||
# Removed specific top/chin extensions to preserve face shape
|
# Expand with extra downward bias toward chin
|
||||||
|
offsets = lower_lip_landmarks - center
|
||||||
|
chin_bias = 1 + (mouth_mask_size / 100.0) * 1.5
|
||||||
|
scale_y = np.where(offsets[:, 1] > 0, expansion_factor * chin_bias, expansion_factor)
|
||||||
|
expanded_landmarks = lower_lip_landmarks.copy()
|
||||||
|
expanded_landmarks[:, 0] = center[0] + offsets[:, 0] * expansion_factor
|
||||||
|
expanded_landmarks[:, 1] = center[1] + offsets[:, 1] * scale_y
|
||||||
|
|
||||||
# Convert back to integer coordinates
|
# Convert back to integer coordinates
|
||||||
expanded_landmarks = expanded_landmarks.astype(np.int32)
|
expanded_landmarks = expanded_landmarks.astype(np.int32)
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
+2
-2
@@ -1,7 +1,7 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
|
|
||||||
# Import the tkinter fix to patch the ScreenChanged error
|
# Import the tkinter fix to patch the ScreenChanged error (module patches Tk on import)
|
||||||
import tkinter_fix
|
import tkinter_fix # noqa: F401
|
||||||
|
|
||||||
import core
|
import core
|
||||||
|
|
||||||
|
|||||||
+1400
-1372
File diff suppressed because it is too large
Load Diff
+49
-7
@@ -30,8 +30,12 @@ def run_ffmpeg(args: List[str]) -> bool:
|
|||||||
try:
|
try:
|
||||||
subprocess.check_output(commands, stderr=subprocess.STDOUT)
|
subprocess.check_output(commands, stderr=subprocess.STDOUT)
|
||||||
return True
|
return True
|
||||||
except Exception:
|
except subprocess.CalledProcessError as error:
|
||||||
pass
|
output = error.output.decode(errors="ignore").strip()
|
||||||
|
if output:
|
||||||
|
print(output)
|
||||||
|
except Exception as error:
|
||||||
|
print(f"ffmpeg execution failed: {error}")
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
@@ -61,19 +65,19 @@ def extract_frames(target_path: str) -> None:
|
|||||||
"""Extract frames with hardware acceleration and optimized settings."""
|
"""Extract frames with hardware acceleration and optimized settings."""
|
||||||
temp_directory_path = get_temp_directory_path(target_path)
|
temp_directory_path = get_temp_directory_path(target_path)
|
||||||
|
|
||||||
# Use hardware-accelerated decoding and optimized pixel format
|
# Write a contiguous image sequence so the later "%04d.png" input pattern
|
||||||
|
# used during encoding can consume every frame reliably.
|
||||||
run_ffmpeg(
|
run_ffmpeg(
|
||||||
[
|
[
|
||||||
"-i", target_path,
|
"-i", target_path,
|
||||||
"-vf", "format=rgb24", # Use video filter for format conversion (faster)
|
"-vf", "format=rgb24", # Use video filter for format conversion (faster)
|
||||||
"-vsync", "0", # Prevent frame duplication
|
"-vsync", "0", # Prevent frame duplication
|
||||||
"-frame_pts", "1", # Preserve frame timing
|
|
||||||
os.path.join(temp_directory_path, "%04d.png"),
|
os.path.join(temp_directory_path, "%04d.png"),
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def create_video(target_path: str, fps: float = 30.0) -> None:
|
def create_video(target_path: str, fps: float = 30.0) -> bool:
|
||||||
"""Create video with hardware-accelerated encoding and optimized settings."""
|
"""Create video with hardware-accelerated encoding and optimized settings."""
|
||||||
temp_output_path = get_temp_output_path(target_path)
|
temp_output_path = get_temp_output_path(target_path)
|
||||||
temp_directory_path = get_temp_directory_path(target_path)
|
temp_directory_path = get_temp_directory_path(target_path)
|
||||||
@@ -182,7 +186,8 @@ def create_video(target_path: str, fps: float = 30.0) -> None:
|
|||||||
"-y",
|
"-y",
|
||||||
temp_output_path,
|
temp_output_path,
|
||||||
]
|
]
|
||||||
run_ffmpeg(ffmpeg_args_fallback)
|
success = run_ffmpeg(ffmpeg_args_fallback)
|
||||||
|
return success and os.path.isfile(temp_output_path)
|
||||||
|
|
||||||
|
|
||||||
def restore_audio(target_path: str, output_path: str) -> None:
|
def restore_audio(target_path: str, output_path: str) -> None:
|
||||||
@@ -257,11 +262,16 @@ def clean_temp(target_path: str) -> None:
|
|||||||
|
|
||||||
|
|
||||||
def has_image_extension(image_path: str) -> bool:
|
def has_image_extension(image_path: str) -> bool:
|
||||||
return image_path.lower().endswith(("png", "jpg", "jpeg"))
|
# splitext so only the real extension counts (e.g. "photo.png.bak" is not
|
||||||
|
# an image); the set is centralized in globals to stay in sync with dialogs.
|
||||||
|
return os.path.splitext(image_path)[1].lower() in modules.globals.IMAGE_EXTENSIONS
|
||||||
|
|
||||||
|
|
||||||
def is_image(image_path: str) -> bool:
|
def is_image(image_path: str) -> bool:
|
||||||
if image_path and os.path.isfile(image_path):
|
if image_path and os.path.isfile(image_path):
|
||||||
|
# Extension check first — Windows mimetypes doesn't always register webp
|
||||||
|
if has_image_extension(image_path):
|
||||||
|
return True
|
||||||
mimetype, _ = mimetypes.guess_type(image_path)
|
mimetype, _ = mimetypes.guess_type(image_path)
|
||||||
return bool(mimetype and mimetype.startswith("image/"))
|
return bool(mimetype and mimetype.startswith("image/"))
|
||||||
return False
|
return False
|
||||||
@@ -309,3 +319,35 @@ def conditional_download(download_directory_path: str, urls: List[str]) -> None:
|
|||||||
|
|
||||||
def resolve_relative_path(path: str) -> str:
|
def resolve_relative_path(path: str) -> str:
|
||||||
return os.path.abspath(os.path.join(os.path.dirname(__file__), path))
|
return os.path.abspath(os.path.join(os.path.dirname(__file__), path))
|
||||||
|
|
||||||
|
|
||||||
|
def get_video_dimensions(target_path: str) -> tuple:
|
||||||
|
"""Get video width and height using ffprobe."""
|
||||||
|
command = [
|
||||||
|
"ffprobe", "-v", "error",
|
||||||
|
"-select_streams", "v:0",
|
||||||
|
"-show_entries", "stream=width,height",
|
||||||
|
"-of", "csv=p=0:s=x",
|
||||||
|
target_path,
|
||||||
|
]
|
||||||
|
output = subprocess.check_output(command).decode().strip()
|
||||||
|
width, height = map(int, output.split("x"))
|
||||||
|
return width, height
|
||||||
|
|
||||||
|
|
||||||
|
def estimate_frame_count(target_path: str, fps: float = None) -> int:
|
||||||
|
"""Estimate total frame count from video duration and fps."""
|
||||||
|
if fps is None:
|
||||||
|
fps = detect_fps(target_path)
|
||||||
|
command = [
|
||||||
|
"ffprobe", "-v", "error",
|
||||||
|
"-show_entries", "format=duration",
|
||||||
|
"-of", "csv=p=0",
|
||||||
|
target_path,
|
||||||
|
]
|
||||||
|
try:
|
||||||
|
output = subprocess.check_output(command).decode().strip()
|
||||||
|
duration = float(output)
|
||||||
|
return int(duration * fps)
|
||||||
|
except Exception:
|
||||||
|
return 0
|
||||||
|
|||||||
+77
-11
@@ -1,5 +1,6 @@
|
|||||||
import cv2
|
import cv2
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
import time
|
||||||
from typing import Optional, Tuple, Callable
|
from typing import Optional, Tuple, Callable
|
||||||
import platform
|
import platform
|
||||||
import threading
|
import threading
|
||||||
@@ -17,6 +18,10 @@ class VideoCapturer:
|
|||||||
self._frame_ready = threading.Event()
|
self._frame_ready = threading.Event()
|
||||||
self.is_running = False
|
self.is_running = False
|
||||||
self.cap = None
|
self.cap = None
|
||||||
|
# Actual values reported by the camera after configuration
|
||||||
|
self.actual_width: int = 0
|
||||||
|
self.actual_height: int = 0
|
||||||
|
self.actual_fps: float = 0.0
|
||||||
|
|
||||||
# Initialize Windows-specific components if on Windows
|
# Initialize Windows-specific components if on Windows
|
||||||
if platform.system() == "Windows":
|
if platform.system() == "Windows":
|
||||||
@@ -32,33 +37,71 @@ class VideoCapturer:
|
|||||||
"""Initialize and start video capture"""
|
"""Initialize and start video capture"""
|
||||||
try:
|
try:
|
||||||
if platform.system() == "Windows":
|
if platform.system() == "Windows":
|
||||||
# Windows-specific capture methods
|
# device_index comes from pygrabber.FilterGraph (DirectShow
|
||||||
|
# enumeration), so open with DSHOW first to preserve mapping.
|
||||||
|
# MSMF and DirectShow enumerate cameras in different orders, so
|
||||||
|
# opening MSMF with a DSHOW index silently selects the wrong
|
||||||
|
# camera. MSMF/ANY remain as fallbacks for cameras DSHOW can't
|
||||||
|
# open.
|
||||||
|
#
|
||||||
|
# Pass codec + resolution + fps as construction params (OpenCV
|
||||||
|
# 4.6+). DSHOW locks the pixel format at open time and ignores
|
||||||
|
# later cap.set(CAP_PROP_FOURCC, ...) — without this, DSHOW
|
||||||
|
# falls back to uncompressed YUYV at 1080p, which is USB-
|
||||||
|
# bandwidth-limited to ~5 fps. Setting MJPG at construction
|
||||||
|
# negotiates compressed frames from the first read.
|
||||||
|
mjpg = cv2.VideoWriter_fourcc(*'MJPG')
|
||||||
|
open_params = [
|
||||||
|
cv2.CAP_PROP_FOURCC, mjpg,
|
||||||
|
cv2.CAP_PROP_FRAME_WIDTH, width,
|
||||||
|
cv2.CAP_PROP_FRAME_HEIGHT, height,
|
||||||
|
cv2.CAP_PROP_FPS, fps,
|
||||||
|
]
|
||||||
capture_methods = [
|
capture_methods = [
|
||||||
(self.device_index, cv2.CAP_DSHOW), # Try DirectShow first
|
(self.device_index, cv2.CAP_DSHOW),
|
||||||
(self.device_index, cv2.CAP_ANY), # Then try default backend
|
(self.device_index, cv2.CAP_MSMF),
|
||||||
(-1, cv2.CAP_ANY), # Try -1 as fallback
|
(self.device_index, cv2.CAP_ANY),
|
||||||
(0, cv2.CAP_ANY), # Finally try 0 without specific backend
|
|
||||||
]
|
]
|
||||||
|
|
||||||
for dev_id, backend in capture_methods:
|
for dev_id, backend in capture_methods:
|
||||||
try:
|
try:
|
||||||
self.cap = cv2.VideoCapture(dev_id, backend)
|
self.cap = cv2.VideoCapture(dev_id, backend, open_params)
|
||||||
if self.cap.isOpened():
|
if self.cap.isOpened():
|
||||||
break
|
break
|
||||||
self.cap.release()
|
self.cap.release()
|
||||||
except Exception:
|
except Exception:
|
||||||
continue
|
continue
|
||||||
|
elif platform.system() == "Linux":
|
||||||
|
self.cap = cv2.VideoCapture(f"/dev/video{self.device_index}")
|
||||||
else:
|
else:
|
||||||
# Unix-like systems (Linux/Mac) capture method
|
|
||||||
self.cap = cv2.VideoCapture(self.device_index)
|
self.cap = cv2.VideoCapture(self.device_index)
|
||||||
|
|
||||||
if not self.cap or not self.cap.isOpened():
|
if not self.cap or not self.cap.isOpened():
|
||||||
raise RuntimeError("Failed to open camera")
|
raise RuntimeError("Failed to open camera")
|
||||||
|
|
||||||
# Configure format
|
# Belt-and-braces: also set via cap.set() for backends that honor
|
||||||
self.cap.set(cv2.CAP_PROP_FRAME_WIDTH, width)
|
# post-open changes (MSMF, V4L2). DSHOW ignores these, but the
|
||||||
self.cap.set(cv2.CAP_PROP_FRAME_HEIGHT, height)
|
# construction params above already handled it.
|
||||||
self.cap.set(cv2.CAP_PROP_FPS, fps)
|
if platform.system() != "Windows":
|
||||||
|
self.cap.set(cv2.CAP_PROP_FOURCC, cv2.VideoWriter_fourcc(*'MJPG'))
|
||||||
|
self.cap.set(cv2.CAP_PROP_FRAME_WIDTH, width)
|
||||||
|
self.cap.set(cv2.CAP_PROP_FRAME_HEIGHT, height)
|
||||||
|
self.cap.set(cv2.CAP_PROP_FPS, fps)
|
||||||
|
|
||||||
|
# Read back resolution (usually reliable)
|
||||||
|
self.actual_width = int(self.cap.get(cv2.CAP_PROP_FRAME_WIDTH))
|
||||||
|
self.actual_height = int(self.cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
|
||||||
|
|
||||||
|
# CAP_PROP_FPS is unreliable on DirectShow — often reports 30
|
||||||
|
# even when the camera delivers 60. Measure empirically by
|
||||||
|
# timing a burst of frames.
|
||||||
|
reported_fps = self.cap.get(cv2.CAP_PROP_FPS)
|
||||||
|
self.actual_fps = self._measure_fps(warmup=10, sample=30,
|
||||||
|
fallback=reported_fps or fps)
|
||||||
|
|
||||||
|
print(f"[VideoCapturer] {self.actual_width}x{self.actual_height} "
|
||||||
|
f"@ {self.actual_fps:.1f}fps (reported={reported_fps:.0f})",
|
||||||
|
flush=True)
|
||||||
|
|
||||||
self.is_running = True
|
self.is_running = True
|
||||||
return True
|
return True
|
||||||
@@ -89,6 +132,29 @@ class VideoCapturer:
|
|||||||
self.is_running = False
|
self.is_running = False
|
||||||
self.cap = None
|
self.cap = None
|
||||||
|
|
||||||
|
def _measure_fps(self, warmup: int = 10, sample: int = 30,
|
||||||
|
fallback: float = 30.0) -> float:
|
||||||
|
"""Read warmup+sample frames and return measured FPS.
|
||||||
|
|
||||||
|
This is more reliable than CAP_PROP_FPS which often lies on
|
||||||
|
DirectShow. Takes ~0.5-1s at startup but gives a ground-truth
|
||||||
|
number for adaptive polling/detection intervals.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
for _ in range(warmup):
|
||||||
|
self.cap.read()
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
for _ in range(sample):
|
||||||
|
ret, _ = self.cap.read()
|
||||||
|
if not ret:
|
||||||
|
return fallback
|
||||||
|
elapsed = time.perf_counter() - t0
|
||||||
|
if elapsed <= 0:
|
||||||
|
return fallback
|
||||||
|
return sample / elapsed
|
||||||
|
except Exception:
|
||||||
|
return fallback
|
||||||
|
|
||||||
def set_frame_callback(self, callback: Callable[[np.ndarray], None]) -> None:
|
def set_frame_callback(self, callback: Callable[[np.ndarray], None]) -> None:
|
||||||
"""Set callback for frame processing"""
|
"""Set callback for frame processing"""
|
||||||
self.frame_callback = callback
|
self.frame_callback = callback
|
||||||
|
|||||||
@@ -0,0 +1,9 @@
|
|||||||
|
[tool.ruff]
|
||||||
|
target-version = "py310"
|
||||||
|
|
||||||
|
[tool.ruff.lint]
|
||||||
|
# Deterministic, low-risk rules enforced in CI. Other rules (F841, E402, F821)
|
||||||
|
# surface real findings but require human judgement to fix safely, so they are
|
||||||
|
# left out of the gate for now. Intentional side-effect imports should be
|
||||||
|
# annotated with `# noqa: F401`.
|
||||||
|
select = ["E701", "E711", "E712", "F401", "F541"]
|
||||||
+17
-15
@@ -1,16 +1,18 @@
|
|||||||
numpy>=1.23.5,<2
|
numpy>=2.0,<3
|
||||||
typing-extensions>=4.8.0
|
typing-extensions>=4.15.0
|
||||||
opencv-python==4.10.0.84
|
opencv-python==4.14.0.94
|
||||||
cv2_enumerate_cameras==1.1.15
|
opencv-python-headless==4.14.0.94
|
||||||
onnx==1.18.0
|
cv2_enumerate_cameras==1.3.3
|
||||||
|
onnx==1.22.0
|
||||||
insightface==0.7.3
|
insightface==0.7.3
|
||||||
psutil==5.9.8
|
psutil==7.2.2
|
||||||
tk==0.1.0
|
PySide6>=6.7,<7
|
||||||
customtkinter==5.2.2
|
pillow==12.3.0
|
||||||
pillow==12.1.1
|
tqdm>=4.66.3
|
||||||
onnxruntime-silicon==1.16.3; sys_platform == 'darwin' and platform_machine == 'arm64'
|
onnxruntime==1.28.0; sys_platform == 'darwin' and platform_machine == 'arm64'
|
||||||
onnxruntime-gpu==1.24.2; sys_platform != 'darwin'
|
onnxruntime==1.23.0; sys_platform == 'darwin' and platform_machine != 'arm64'
|
||||||
tensorflow; sys_platform != 'darwin'
|
onnxruntime-gpu==1.26.0; sys_platform != 'darwin'
|
||||||
opennsfw2==0.10.2
|
opennsfw2==0.18.0
|
||||||
protobuf==4.25.1
|
keras>=3.0.0
|
||||||
pygrabber
|
protobuf>=6.33.5,<8
|
||||||
|
pygrabber; sys_platform == 'win32'
|
||||||
|
|||||||
@@ -1,7 +1,97 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
|
|
||||||
# Import the tkinter fix to patch the ScreenChanged error
|
import os
|
||||||
import tkinter_fix
|
import sys
|
||||||
|
|
||||||
|
# Add the project root to PATH so bundled ffmpeg/ffprobe are found
|
||||||
|
project_root = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
os.environ["PATH"] = project_root + os.pathsep + os.environ.get("PATH", "")
|
||||||
|
|
||||||
|
# On Windows, register NVIDIA CUDA DLL directories so onnxruntime-gpu can
|
||||||
|
# find cuDNN/cublas. Python 3.8+ ignores PATH for extension-module native deps —
|
||||||
|
# os.add_dll_directory() is required. Also keep PATH for child processes/ffmpeg.
|
||||||
|
if sys.platform == "win32":
|
||||||
|
_site_packages = os.path.join(sys.prefix, "Lib", "site-packages")
|
||||||
|
_venv_site_packages = os.path.join(project_root, "venv", "Lib", "site-packages")
|
||||||
|
for _sp in (_site_packages, _venv_site_packages):
|
||||||
|
_candidate_dirs = []
|
||||||
|
_torch_lib = os.path.join(_sp, "torch", "lib")
|
||||||
|
if os.path.isdir(_torch_lib):
|
||||||
|
_candidate_dirs.append(_torch_lib)
|
||||||
|
_nvidia_dir = os.path.join(_sp, "nvidia")
|
||||||
|
if os.path.isdir(_nvidia_dir):
|
||||||
|
for _pkg in os.listdir(_nvidia_dir):
|
||||||
|
_bin_dir = os.path.join(_nvidia_dir, _pkg, "bin")
|
||||||
|
if os.path.isdir(_bin_dir):
|
||||||
|
_candidate_dirs.append(_bin_dir)
|
||||||
|
for _d in _candidate_dirs:
|
||||||
|
os.environ["PATH"] = _d + os.pathsep + os.environ["PATH"]
|
||||||
|
try:
|
||||||
|
os.add_dll_directory(_d)
|
||||||
|
except (OSError, AttributeError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
# On Windows, register OpenVINO DLL directories so onnxruntime's
|
||||||
|
# OpenVINOExecutionProvider can find openvino.dll. This must happen
|
||||||
|
# before any ONNX InferenceSession is created. Failure is non-fatal:
|
||||||
|
# OpenVINO simply isn't installed, and onnxruntime will fall back to CPU.
|
||||||
|
try:
|
||||||
|
from onnxruntime.tools.add_openvino_win_libs import ( # type: ignore[import-untyped] # noqa: E501
|
||||||
|
add_openvino_libs_to_path,
|
||||||
|
)
|
||||||
|
add_openvino_libs_to_path()
|
||||||
|
except ImportError:
|
||||||
|
# onnxruntime build without the OpenVINO tooling module — no-op.
|
||||||
|
pass
|
||||||
|
except FileNotFoundError:
|
||||||
|
# OpenVINO site-packages dir absent — no-op.
|
||||||
|
pass
|
||||||
|
except SystemExit as exc:
|
||||||
|
# add_openvino_libs_to_path() calls sys.exit() when OpenVINO libs
|
||||||
|
# can't be located (e.g. OPENVINO_LIB_PATHS unset). Log the message
|
||||||
|
# it raised with so the failure is visible, but keep startup alive.
|
||||||
|
print(
|
||||||
|
f"[startup] OpenVINO DLL registration skipped: {exc}",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
# On Linux, pre-load NVIDIA shared libraries (cuDNN, cuBLAS, nvrtc...) shipped
|
||||||
|
# inside the venv via pip wheels (nvidia-cudnn-cu12, etc.). LD_LIBRARY_PATH
|
||||||
|
# cannot be set after Python starts, so we use ctypes.CDLL with RTLD_GLOBAL
|
||||||
|
# instead. This makes symbols available to onnxruntime when it dlopens its
|
||||||
|
# CUDA provider.
|
||||||
|
if sys.platform.startswith("linux"):
|
||||||
|
import ctypes
|
||||||
|
import glob
|
||||||
|
_py_lib = f"python{sys.version_info.major}.{sys.version_info.minor}"
|
||||||
|
_site_packages_candidates = [
|
||||||
|
os.path.join(project_root, "venv", "lib", _py_lib, "site-packages"),
|
||||||
|
os.path.join(sys.prefix, "lib", _py_lib, "site-packages"),
|
||||||
|
]
|
||||||
|
for _sp in _site_packages_candidates:
|
||||||
|
_nvidia_dir = os.path.join(_sp, "nvidia")
|
||||||
|
if not os.path.isdir(_nvidia_dir):
|
||||||
|
continue
|
||||||
|
for _pkg in os.listdir(_nvidia_dir):
|
||||||
|
_lib_dir = os.path.join(_nvidia_dir, _pkg, "lib")
|
||||||
|
if not os.path.isdir(_lib_dir):
|
||||||
|
continue
|
||||||
|
# Also expose the directory to child processes, without
|
||||||
|
# duplicating an entry that is already present.
|
||||||
|
_ldp = os.environ.get("LD_LIBRARY_PATH", "")
|
||||||
|
if _lib_dir not in _ldp.split(os.pathsep):
|
||||||
|
os.environ["LD_LIBRARY_PATH"] = (
|
||||||
|
_lib_dir + (os.pathsep + _ldp if _ldp else "")
|
||||||
|
)
|
||||||
|
for _so in sorted(glob.glob(os.path.join(_lib_dir, "lib*.so*"))):
|
||||||
|
try:
|
||||||
|
ctypes.CDLL(_so, mode=ctypes.RTLD_GLOBAL)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
break
|
||||||
|
|
||||||
|
from modules import platform_info
|
||||||
|
platform_info.print_banner()
|
||||||
|
|
||||||
from modules import core
|
from modules import core
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,137 @@
|
|||||||
|
import importlib
|
||||||
|
import sys
|
||||||
|
import types
|
||||||
|
import unittest
|
||||||
|
from contextlib import contextmanager
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
|
||||||
|
@contextmanager
|
||||||
|
def _patched_core_import_stubs(calls, pipe_result=False):
|
||||||
|
class Processor:
|
||||||
|
NAME = "test_processor"
|
||||||
|
|
||||||
|
def pre_start(self):
|
||||||
|
return True
|
||||||
|
|
||||||
|
def pre_check(self):
|
||||||
|
return True
|
||||||
|
|
||||||
|
def process_image(self, *_args, **_kwargs):
|
||||||
|
raise AssertionError("image path should not be used")
|
||||||
|
|
||||||
|
def process_video(self, source_path, frame_paths):
|
||||||
|
calls.append(("process_video", source_path, tuple(frame_paths)))
|
||||||
|
|
||||||
|
stubs = {
|
||||||
|
"cv2": types.SimpleNamespace(
|
||||||
|
IMREAD_COLOR=1,
|
||||||
|
imdecode=lambda *_args, **_kwargs: None,
|
||||||
|
imencode=lambda *_args, **_kwargs: (
|
||||||
|
True,
|
||||||
|
types.SimpleNamespace(tofile=lambda *_a, **_k: None),
|
||||||
|
),
|
||||||
|
),
|
||||||
|
"numpy": types.SimpleNamespace(uint8=object, fromfile=lambda *_args, **_kwargs: b""),
|
||||||
|
"torch": types.SimpleNamespace(
|
||||||
|
cuda=types.SimpleNamespace(empty_cache=lambda: None)
|
||||||
|
),
|
||||||
|
"onnxruntime": types.SimpleNamespace(
|
||||||
|
get_available_providers=lambda: ["CPUExecutionProvider"]
|
||||||
|
),
|
||||||
|
"tensorflow": types.SimpleNamespace(),
|
||||||
|
"modules.metadata": types.SimpleNamespace(name="Deep-Live-Cam", version="test"),
|
||||||
|
"modules.ui": types.SimpleNamespace(
|
||||||
|
check_and_ignore_nsfw=lambda *_args, **_kwargs: False,
|
||||||
|
update_status=lambda *_args, **_kwargs: None,
|
||||||
|
init=lambda *_args, **_kwargs: types.SimpleNamespace(mainloop=lambda: None),
|
||||||
|
),
|
||||||
|
"modules.processors.frame.core": types.SimpleNamespace(
|
||||||
|
get_frame_processors_modules=lambda _names: [Processor()],
|
||||||
|
process_video_in_memory=lambda *_args, **_kwargs: calls.append(("pipe",))
|
||||||
|
or pipe_result,
|
||||||
|
),
|
||||||
|
"modules.utilities": types.SimpleNamespace(
|
||||||
|
has_image_extension=lambda _path: False,
|
||||||
|
is_image=lambda _path: False,
|
||||||
|
is_video=lambda _path: True,
|
||||||
|
detect_fps=lambda _path: 24.0,
|
||||||
|
create_video=lambda target_path, fps: calls.append(
|
||||||
|
("create_video", target_path, fps)
|
||||||
|
)
|
||||||
|
or True,
|
||||||
|
extract_frames=lambda target_path: calls.append(
|
||||||
|
("extract_frames", target_path)
|
||||||
|
),
|
||||||
|
get_temp_frame_paths=lambda target_path: [f"{target_path}/0001.png"],
|
||||||
|
restore_audio=lambda *_args, **_kwargs: calls.append(("restore_audio",)),
|
||||||
|
create_temp=lambda target_path: calls.append(("create_temp", target_path)),
|
||||||
|
move_temp=lambda target_path, output_path: calls.append(
|
||||||
|
("move_temp", target_path, output_path)
|
||||||
|
),
|
||||||
|
clean_temp=lambda target_path: calls.append(("clean_temp", target_path)),
|
||||||
|
normalize_output_path=lambda _source, _target, output: output,
|
||||||
|
),
|
||||||
|
}
|
||||||
|
with patch.dict(sys.modules, stubs, clear=False):
|
||||||
|
sys.modules.pop("modules.core", None)
|
||||||
|
yield importlib.import_module("modules.core")
|
||||||
|
sys.modules.pop("modules.core", None)
|
||||||
|
|
||||||
|
|
||||||
|
def _configure_video_run(core, *, map_faces):
|
||||||
|
core.modules.globals.source_path = "source.jpg"
|
||||||
|
core.modules.globals.target_path = "target.mp4"
|
||||||
|
core.modules.globals.output_path = "output.mp4"
|
||||||
|
core.modules.globals.frame_processors = ["face_swapper"]
|
||||||
|
core.modules.globals.headless = True
|
||||||
|
core.modules.globals.keep_fps = False
|
||||||
|
core.modules.globals.keep_audio = False
|
||||||
|
core.modules.globals.keep_frames = False
|
||||||
|
core.modules.globals.map_faces = map_faces
|
||||||
|
core.modules.globals.nsfw_filter = False
|
||||||
|
core.modules.globals.execution_threads = 1
|
||||||
|
core.modules.globals.execution_providers = ["CPUExecutionProvider"]
|
||||||
|
core.modules.globals.max_memory = None
|
||||||
|
|
||||||
|
|
||||||
|
class MapFacesFallbackTests(unittest.TestCase):
|
||||||
|
def test_map_faces_disk_fallback_extracts_frames_before_processing(self):
|
||||||
|
calls = []
|
||||||
|
with _patched_core_import_stubs(calls, pipe_result=False) as core:
|
||||||
|
_configure_video_run(core, map_faces=True)
|
||||||
|
|
||||||
|
with patch.object(core.os.path, "isfile", return_value=True):
|
||||||
|
core.start()
|
||||||
|
|
||||||
|
self.assertNotIn(("pipe",), calls)
|
||||||
|
self.assertIn(("create_temp", "target.mp4"), calls)
|
||||||
|
self.assertIn(("extract_frames", "target.mp4"), calls)
|
||||||
|
self.assertIn(("process_video", "source.jpg", ("target.mp4/0001.png",)), calls)
|
||||||
|
self.assertIn(("create_video", "target.mp4", 30.0), calls)
|
||||||
|
self.assertIn(("move_temp", "target.mp4", "output.mp4"), calls)
|
||||||
|
|
||||||
|
step_indices = {}
|
||||||
|
for index, call in enumerate(calls):
|
||||||
|
step_indices.setdefault(call[0], index)
|
||||||
|
|
||||||
|
self.assertLess(step_indices["create_temp"], step_indices["extract_frames"])
|
||||||
|
self.assertLess(step_indices["extract_frames"], step_indices["process_video"])
|
||||||
|
self.assertLess(step_indices["process_video"], step_indices["create_video"])
|
||||||
|
self.assertLess(step_indices["create_video"], step_indices["move_temp"])
|
||||||
|
|
||||||
|
def test_non_map_faces_pipe_success_does_not_extract_frames(self):
|
||||||
|
calls = []
|
||||||
|
with _patched_core_import_stubs(calls, pipe_result=True) as core:
|
||||||
|
_configure_video_run(core, map_faces=False)
|
||||||
|
|
||||||
|
with patch.object(core.os.path, "isfile", return_value=True):
|
||||||
|
core.start()
|
||||||
|
|
||||||
|
self.assertIn(("pipe",), calls)
|
||||||
|
self.assertNotIn(("extract_frames", "target.mp4"), calls)
|
||||||
|
self.assertNotIn(("process_video", "source.jpg", ("target.mp4/0001.png",)), calls)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,97 @@
|
|||||||
|
import importlib
|
||||||
|
import sys
|
||||||
|
import types
|
||||||
|
import unittest
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
|
||||||
|
def _install_import_stubs():
|
||||||
|
sys.modules.setdefault(
|
||||||
|
"insightface",
|
||||||
|
types.SimpleNamespace(app=types.SimpleNamespace(FaceAnalysis=object)),
|
||||||
|
)
|
||||||
|
sys.modules.setdefault(
|
||||||
|
"cv2",
|
||||||
|
types.SimpleNamespace(
|
||||||
|
IMREAD_COLOR=1,
|
||||||
|
imread=lambda *_args, **_kwargs: None,
|
||||||
|
imdecode=lambda *_args, **_kwargs: None,
|
||||||
|
imencode=lambda *_args, **_kwargs: (
|
||||||
|
True,
|
||||||
|
types.SimpleNamespace(tofile=lambda *_a, **_k: None),
|
||||||
|
),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
sys.modules.setdefault(
|
||||||
|
"numpy",
|
||||||
|
types.SimpleNamespace(uint8=object, fromfile=lambda *_args, **_kwargs: b""),
|
||||||
|
)
|
||||||
|
sys.modules.setdefault(
|
||||||
|
"tqdm",
|
||||||
|
types.SimpleNamespace(tqdm=lambda iterable, **_kwargs: iterable),
|
||||||
|
)
|
||||||
|
sys.modules["modules.typing"] = types.SimpleNamespace(Frame=object)
|
||||||
|
sys.modules["modules.cluster_analysis"] = types.SimpleNamespace(
|
||||||
|
find_cluster_centroids=lambda *args, **kwargs: [],
|
||||||
|
find_closest_centroid=lambda *args, **kwargs: (0, None),
|
||||||
|
)
|
||||||
|
sys.modules["modules.utilities"] = types.SimpleNamespace(
|
||||||
|
get_temp_directory_path=lambda path: path,
|
||||||
|
create_temp=lambda path: None,
|
||||||
|
extract_frames=lambda path: None,
|
||||||
|
clean_temp=lambda path: None,
|
||||||
|
get_temp_frame_paths=lambda path: [],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _load_face_analyser():
|
||||||
|
_install_import_stubs()
|
||||||
|
sys.modules.pop("modules.face_analyser", None)
|
||||||
|
return importlib.import_module("modules.face_analyser")
|
||||||
|
|
||||||
|
|
||||||
|
class Face:
|
||||||
|
def __init__(self, left):
|
||||||
|
self.bbox = [left, 0, 10, 10]
|
||||||
|
|
||||||
|
|
||||||
|
class GetOneFaceTests(unittest.TestCase):
|
||||||
|
def test_uses_supplied_detected_faces_without_reanalysing_frame(self):
|
||||||
|
face_analyser = _load_face_analyser()
|
||||||
|
right = Face(20)
|
||||||
|
left = Face(5)
|
||||||
|
|
||||||
|
with patch.object(
|
||||||
|
face_analyser,
|
||||||
|
"_analyse_faces",
|
||||||
|
side_effect=AssertionError("should not analyse"),
|
||||||
|
):
|
||||||
|
self.assertIs(face_analyser.get_one_face("frame", [right, left]), left)
|
||||||
|
|
||||||
|
def test_supplied_empty_detected_faces_returns_none(self):
|
||||||
|
face_analyser = _load_face_analyser()
|
||||||
|
|
||||||
|
with patch.object(
|
||||||
|
face_analyser,
|
||||||
|
"_analyse_faces",
|
||||||
|
side_effect=AssertionError("should not analyse"),
|
||||||
|
):
|
||||||
|
self.assertIsNone(face_analyser.get_one_face("frame", []))
|
||||||
|
|
||||||
|
def test_without_supplied_faces_preserves_existing_detection_path(self):
|
||||||
|
face_analyser = _load_face_analyser()
|
||||||
|
right = Face(30)
|
||||||
|
left = Face(3)
|
||||||
|
|
||||||
|
with patch.object(face_analyser, "_is_dml", return_value=False), patch.object(
|
||||||
|
face_analyser,
|
||||||
|
"_analyse_faces",
|
||||||
|
return_value=[right, left],
|
||||||
|
) as analyse:
|
||||||
|
self.assertIs(face_analyser.get_one_face("frame"), left)
|
||||||
|
|
||||||
|
analyse.assert_called_once_with("frame")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -1,3 +1,6 @@
|
|||||||
|
import os
|
||||||
|
os.environ.setdefault('TK_SILENCE_DEPRECATION', '1')
|
||||||
|
|
||||||
import tkinter
|
import tkinter
|
||||||
|
|
||||||
# Only needs to be imported once at the beginning of the application
|
# Only needs to be imported once at the beginning of the application
|
||||||
|
|||||||
Reference in New Issue
Block a user