mirror of
https://github.com/facefusion/facefusion.git
synced 2026-07-30 22:07:23 +02:00
Read the multi-window target frames via ffmpeg/ffprobe instead of cv2
video_manager built its video reader on cv2.VideoCapture, which cannot decode AV1 with the bundled opencv wheel on Linux, so the multi-window target pack (read_video_chunk -> select_video_frames) came back empty and frames passed through unswapped. Replace the reader with a persistent ffmpeg rawvideo pipe whose instance lives in the pool store, mirroring the cv2.VideoCapture lifecycle: probe width/height/fps/frame_total via ffprobe, spawn an ffmpeg pipe seeked to the chunk start, read frames sequentially, restart on a backward seek. All ffmpeg/ffprobe command construction goes through the builders; the single-frame cv2 path is left untouched (scope is the multi-window reader). ffprobe.py / ffprobe_builder.py follow the v4 template. AV1 now swaps correctly at ~20 fps (vs the ~11 fps temp-frame variant); H.264 stays correct at ~18 fps. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
3f81a8a784
commit
d30f2e2f8e
@@ -83,6 +83,14 @@ def unsafe_concat() -> List[Command]:
|
|||||||
return [ '-f', 'concat', '-safe', '0' ]
|
return [ '-f', 'concat', '-safe', '0' ]
|
||||||
|
|
||||||
|
|
||||||
|
def set_input_seek(seek_time : float) -> List[Command]:
|
||||||
|
return [ '-ss', str(seek_time) ]
|
||||||
|
|
||||||
|
|
||||||
|
def set_output_format(output_format : str) -> List[Command]:
|
||||||
|
return [ '-f', output_format ]
|
||||||
|
|
||||||
|
|
||||||
def enforce_pixel_format(pixel_format : str) -> List[Command]:
|
def enforce_pixel_format(pixel_format : str) -> List[Command]:
|
||||||
return [ '-pix_fmt', pixel_format ]
|
return [ '-pix_fmt', pixel_format ]
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,79 @@
|
|||||||
|
import subprocess
|
||||||
|
from typing import Dict, List
|
||||||
|
|
||||||
|
from facefusion import ffprobe_builder
|
||||||
|
from facefusion.types import Buffer, Command, Fps, VideoMetadata
|
||||||
|
|
||||||
|
|
||||||
|
def run_ffprobe(commands : List[Command]) -> subprocess.Popen[Buffer]:
|
||||||
|
commands = ffprobe_builder.run(commands)
|
||||||
|
return subprocess.Popen(commands, stderr = subprocess.PIPE, stdout = subprocess.PIPE)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_entries(output : Buffer) -> Dict[str, str]:
|
||||||
|
media_entries = {}
|
||||||
|
|
||||||
|
if output:
|
||||||
|
lines = output.decode().strip().splitlines()
|
||||||
|
|
||||||
|
for line in lines:
|
||||||
|
if '=' in line:
|
||||||
|
key, value = line.split('=', 1)
|
||||||
|
media_entries[key] = value
|
||||||
|
|
||||||
|
return media_entries
|
||||||
|
|
||||||
|
|
||||||
|
def probe_entries(media_path : str, entries : List[str]) -> Dict[str, str]:
|
||||||
|
commands = ffprobe_builder.chain(
|
||||||
|
ffprobe_builder.show_entries(entries),
|
||||||
|
ffprobe_builder.format_to_key_value(),
|
||||||
|
ffprobe_builder.set_input(media_path)
|
||||||
|
)
|
||||||
|
output, _ = run_ffprobe(commands).communicate()
|
||||||
|
|
||||||
|
return parse_entries(output)
|
||||||
|
|
||||||
|
|
||||||
|
def probe_video_entries(video_path : str, entries : List[str]) -> Dict[str, str]:
|
||||||
|
commands = ffprobe_builder.chain(
|
||||||
|
ffprobe_builder.select_video_stream(),
|
||||||
|
ffprobe_builder.show_entries(entries),
|
||||||
|
ffprobe_builder.format_to_key_value(),
|
||||||
|
ffprobe_builder.set_input(video_path)
|
||||||
|
)
|
||||||
|
output, _ = run_ffprobe(commands).communicate()
|
||||||
|
|
||||||
|
return parse_entries(output)
|
||||||
|
|
||||||
|
|
||||||
|
def extract_video_metadata(video_path : str) -> VideoMetadata:
|
||||||
|
video_entries = probe_video_entries(video_path, [ 'duration', 'width', 'height', 'r_frame_rate', 'bit_rate' ])
|
||||||
|
|
||||||
|
duration = float(video_entries.get('duration'))
|
||||||
|
fps = extract_video_fps(video_entries.get('r_frame_rate'))
|
||||||
|
frame_total = int(duration * fps)
|
||||||
|
width = int(video_entries.get('width'))
|
||||||
|
height = int(video_entries.get('height'))
|
||||||
|
bit_rate = int(video_entries.get('bit_rate'))
|
||||||
|
|
||||||
|
video_metadata : VideoMetadata =\
|
||||||
|
{
|
||||||
|
'duration' : duration,
|
||||||
|
'frame_total' : frame_total,
|
||||||
|
'fps' : fps,
|
||||||
|
'resolution' : (width, height),
|
||||||
|
'bit_rate' : bit_rate
|
||||||
|
}
|
||||||
|
|
||||||
|
return video_metadata
|
||||||
|
|
||||||
|
|
||||||
|
def extract_video_fps(frame_rate : str) -> Fps:
|
||||||
|
if frame_rate and '/' in frame_rate:
|
||||||
|
numerator, denominator = frame_rate.split('/')
|
||||||
|
|
||||||
|
if int(numerator) and int(denominator):
|
||||||
|
return int(numerator) / int(denominator)
|
||||||
|
|
||||||
|
return 0.0
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
import itertools
|
||||||
|
import shutil
|
||||||
|
from typing import List
|
||||||
|
|
||||||
|
from facefusion.types import Command
|
||||||
|
|
||||||
|
|
||||||
|
def run(commands : List[Command]) -> List[Command]:
|
||||||
|
return [ shutil.which('ffprobe'), '-loglevel', 'error' ] + commands
|
||||||
|
|
||||||
|
|
||||||
|
def chain(*commands : List[Command]) -> List[Command]:
|
||||||
|
return list(itertools.chain(*commands))
|
||||||
|
|
||||||
|
|
||||||
|
def select_video_stream() -> List[Command]:
|
||||||
|
return [ '-select_streams', 'v:0' ]
|
||||||
|
|
||||||
|
|
||||||
|
def show_entries(entries : List[str]) -> List[Command]:
|
||||||
|
return [ '-show_entries', 'stream=' + ','.join(entries) ]
|
||||||
|
|
||||||
|
|
||||||
|
def format_to_value() -> List[Command]:
|
||||||
|
return [ '-of', 'default=noprint_wrappers=1:nokey=1' ]
|
||||||
|
|
||||||
|
|
||||||
|
def format_to_key_value() -> List[Command]:
|
||||||
|
return [ '-of', 'default=noprint_wrappers=1' ]
|
||||||
|
|
||||||
|
|
||||||
|
def set_input(input_path : str) -> List[Command]:
|
||||||
|
return [ '-i', input_path ]
|
||||||
+29
-5
@@ -1,3 +1,4 @@
|
|||||||
|
import subprocess
|
||||||
from collections import namedtuple
|
from collections import namedtuple
|
||||||
from threading import Lock
|
from threading import Lock
|
||||||
from typing import Any, Callable, Dict, List, Literal, NotRequired, Optional, Tuple, TypeAlias, TypedDict
|
from typing import Any, Callable, Dict, List, Literal, NotRequired, Optional, Tuple, TypeAlias, TypedDict
|
||||||
@@ -66,11 +67,6 @@ LocalePoolSet : TypeAlias = Dict[str, Locales]
|
|||||||
VideoCaptureSet : TypeAlias = Dict[str, cv2.VideoCapture]
|
VideoCaptureSet : TypeAlias = Dict[str, cv2.VideoCapture]
|
||||||
VideoWriterSet : TypeAlias = Dict[str, cv2.VideoWriter]
|
VideoWriterSet : TypeAlias = Dict[str, cv2.VideoWriter]
|
||||||
CameraCaptureSet : TypeAlias = Dict[str, cv2.VideoCapture]
|
CameraCaptureSet : TypeAlias = Dict[str, cv2.VideoCapture]
|
||||||
VideoPoolSet = TypedDict('VideoPoolSet',
|
|
||||||
{
|
|
||||||
'capture' : VideoCaptureSet,
|
|
||||||
'writer' : VideoWriterSet
|
|
||||||
})
|
|
||||||
CameraPoolSet = TypedDict('CameraPoolSet',
|
CameraPoolSet = TypedDict('CameraPoolSet',
|
||||||
{
|
{
|
||||||
'capture' : CameraCaptureSet
|
'capture' : CameraCaptureSet
|
||||||
@@ -97,11 +93,39 @@ VoiceChunk : TypeAlias = NDArray[Any]
|
|||||||
|
|
||||||
Fps : TypeAlias = float
|
Fps : TypeAlias = float
|
||||||
Duration : TypeAlias = float
|
Duration : TypeAlias = float
|
||||||
|
|
||||||
|
Buffer : TypeAlias = bytes
|
||||||
|
BitRate : TypeAlias = int
|
||||||
|
VideoReader = TypedDict('VideoReader',
|
||||||
|
{
|
||||||
|
'process' : subprocess.Popen,
|
||||||
|
'video_path' : str,
|
||||||
|
'width' : int,
|
||||||
|
'height' : int,
|
||||||
|
'fps' : Fps,
|
||||||
|
'frame_total' : int,
|
||||||
|
'position' : int
|
||||||
|
})
|
||||||
|
VideoReaderSet : TypeAlias = Dict[str, VideoReader]
|
||||||
|
VideoPoolSet = TypedDict('VideoPoolSet',
|
||||||
|
{
|
||||||
|
'capture' : VideoCaptureSet,
|
||||||
|
'writer' : VideoWriterSet,
|
||||||
|
'reader' : VideoReaderSet
|
||||||
|
})
|
||||||
Color : TypeAlias = Tuple[int, int, int, int]
|
Color : TypeAlias = Tuple[int, int, int, int]
|
||||||
Padding : TypeAlias = Tuple[int, int, int, int]
|
Padding : TypeAlias = Tuple[int, int, int, int]
|
||||||
Margin : TypeAlias = Tuple[int, int, int, int]
|
Margin : TypeAlias = Tuple[int, int, int, int]
|
||||||
Orientation = Literal['landscape', 'portrait']
|
Orientation = Literal['landscape', 'portrait']
|
||||||
Resolution : TypeAlias = Tuple[int, int]
|
Resolution : TypeAlias = Tuple[int, int]
|
||||||
|
VideoMetadata = TypedDict('VideoMetadata',
|
||||||
|
{
|
||||||
|
'duration' : Duration,
|
||||||
|
'frame_total' : int,
|
||||||
|
'fps' : Fps,
|
||||||
|
'resolution' : Resolution,
|
||||||
|
'bit_rate' : BitRate
|
||||||
|
})
|
||||||
|
|
||||||
ProcessState = Literal['checking', 'processing', 'stopping', 'pending']
|
ProcessState = Literal['checking', 'processing', 'stopping', 'pending']
|
||||||
Args : TypeAlias = Dict[str, Any]
|
Args : TypeAlias = Dict[str, Any]
|
||||||
|
|||||||
@@ -1,11 +1,18 @@
|
|||||||
import cv2
|
import subprocess
|
||||||
|
from typing import Optional, Tuple
|
||||||
|
|
||||||
from facefusion.types import VideoPoolSet
|
import cv2
|
||||||
|
import numpy
|
||||||
|
|
||||||
|
from facefusion import ffmpeg_builder
|
||||||
|
from facefusion.ffprobe import extract_video_metadata
|
||||||
|
from facefusion.types import Fps, VideoPoolSet, VideoReader, VisionFrame
|
||||||
|
|
||||||
VIDEO_POOL_SET : VideoPoolSet =\
|
VIDEO_POOL_SET : VideoPoolSet =\
|
||||||
{
|
{
|
||||||
'capture': {},
|
'capture': {},
|
||||||
'writer': {}
|
'writer': {},
|
||||||
|
'reader': {}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -25,6 +32,55 @@ def conditional_set_video_frame_position(video_capture : cv2.VideoCapture, frame
|
|||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def create_video_reader_process(video_path : str, frame_position : int, video_fps : Fps) -> subprocess.Popen[bytes]:
|
||||||
|
commands = ffmpeg_builder.chain(
|
||||||
|
ffmpeg_builder.set_input_seek(frame_position / video_fps),
|
||||||
|
ffmpeg_builder.set_input(video_path),
|
||||||
|
ffmpeg_builder.prevent_frame_drop(),
|
||||||
|
ffmpeg_builder.enforce_pixel_format('bgr24'),
|
||||||
|
ffmpeg_builder.set_output_format('rawvideo'),
|
||||||
|
ffmpeg_builder.cast_stream()
|
||||||
|
)
|
||||||
|
commands = ffmpeg_builder.run(commands)
|
||||||
|
return subprocess.Popen(commands, stdout = subprocess.PIPE, stderr = subprocess.DEVNULL)
|
||||||
|
|
||||||
|
|
||||||
|
def get_video_reader(video_path : str) -> VideoReader:
|
||||||
|
if video_path not in VIDEO_POOL_SET.get('reader'):
|
||||||
|
video_metadata = extract_video_metadata(video_path)
|
||||||
|
width, height = video_metadata.get('resolution')
|
||||||
|
|
||||||
|
VIDEO_POOL_SET['reader'][video_path] =\
|
||||||
|
{
|
||||||
|
'process': create_video_reader_process(video_path, 0, video_metadata.get('fps')),
|
||||||
|
'video_path': video_path,
|
||||||
|
'width': width,
|
||||||
|
'height': height,
|
||||||
|
'fps': video_metadata.get('fps'),
|
||||||
|
'frame_total': video_metadata.get('frame_total'),
|
||||||
|
'position': 0
|
||||||
|
}
|
||||||
|
|
||||||
|
return VIDEO_POOL_SET.get('reader').get(video_path)
|
||||||
|
|
||||||
|
|
||||||
|
def conditional_set_video_reader_position(video_reader : VideoReader, frame_position : int) -> None:
|
||||||
|
if not video_reader.get('position') == frame_position:
|
||||||
|
video_reader.get('process').terminate()
|
||||||
|
video_reader['process'] = create_video_reader_process(video_reader.get('video_path'), frame_position, video_reader.get('fps'))
|
||||||
|
video_reader['position'] = frame_position
|
||||||
|
|
||||||
|
|
||||||
|
def read_video_reader_frame(video_reader : VideoReader) -> Tuple[bool, Optional[VisionFrame]]:
|
||||||
|
frame_size = video_reader.get('width') * video_reader.get('height') * 3
|
||||||
|
frame_buffer = video_reader.get('process').stdout.read(frame_size)
|
||||||
|
|
||||||
|
if len(frame_buffer) == frame_size:
|
||||||
|
video_reader['position'] = video_reader.get('position') + 1
|
||||||
|
return True, numpy.frombuffer(frame_buffer, numpy.uint8).reshape(video_reader.get('height'), video_reader.get('width'), 3).copy()
|
||||||
|
return False, None
|
||||||
|
|
||||||
|
|
||||||
def get_video_writer(video_path : str) -> cv2.VideoWriter:
|
def get_video_writer(video_path : str) -> cv2.VideoWriter:
|
||||||
if video_path not in VIDEO_POOL_SET.get('writer'):
|
if video_path not in VIDEO_POOL_SET.get('writer'):
|
||||||
video_writer = cv2.VideoWriter()
|
video_writer = cv2.VideoWriter()
|
||||||
@@ -42,5 +98,9 @@ def clear_video_pool() -> None:
|
|||||||
for video_writer in VIDEO_POOL_SET.get('writer').values():
|
for video_writer in VIDEO_POOL_SET.get('writer').values():
|
||||||
video_writer.release()
|
video_writer.release()
|
||||||
|
|
||||||
|
for video_reader in VIDEO_POOL_SET.get('reader').values():
|
||||||
|
video_reader.get('process').terminate()
|
||||||
|
|
||||||
VIDEO_POOL_SET['capture'].clear()
|
VIDEO_POOL_SET['capture'].clear()
|
||||||
VIDEO_POOL_SET['writer'].clear()
|
VIDEO_POOL_SET['writer'].clear()
|
||||||
|
VIDEO_POOL_SET['reader'].clear()
|
||||||
|
|||||||
+10
-7
@@ -10,7 +10,7 @@ from facefusion.common_helper import is_windows
|
|||||||
from facefusion.filesystem import get_file_extension, is_image, is_video
|
from facefusion.filesystem import get_file_extension, is_image, is_video
|
||||||
from facefusion.thread_helper import thread_lock, thread_semaphore
|
from facefusion.thread_helper import thread_lock, thread_semaphore
|
||||||
from facefusion.types import ColorMode, Duration, Fps, Mask, Orientation, Resolution, Scale, VisionFrame
|
from facefusion.types import ColorMode, Duration, Fps, Mask, Orientation, Resolution, Scale, VisionFrame
|
||||||
from facefusion.video_manager import conditional_set_video_frame_position, get_video_capture
|
from facefusion.video_manager import conditional_set_video_frame_position, conditional_set_video_reader_position, get_video_capture, get_video_reader, read_video_reader_frame
|
||||||
|
|
||||||
|
|
||||||
def read_static_images(image_paths : List[str], color_mode : ColorMode = 'rgb') -> List[VisionFrame]:
|
def read_static_images(image_paths : List[str], color_mode : ColorMode = 'rgb') -> List[VisionFrame]:
|
||||||
@@ -102,17 +102,20 @@ def read_video_chunk(video_path : str, chunk_number : int, chunk_size : int) ->
|
|||||||
video_frame_chunk = {}
|
video_frame_chunk = {}
|
||||||
|
|
||||||
if is_video(video_path) and chunk_number > -1:
|
if is_video(video_path) and chunk_number > -1:
|
||||||
video_capture = get_video_capture(video_path)
|
video_reader = get_video_reader(video_path)
|
||||||
|
|
||||||
if video_capture and video_capture.isOpened():
|
if video_reader:
|
||||||
video_frame_total = int(video_capture.get(cv2.CAP_PROP_FRAME_COUNT))
|
|
||||||
video_frame_position = chunk_number * chunk_size
|
video_frame_position = chunk_number * chunk_size
|
||||||
|
video_frame_end = video_frame_position + chunk_size
|
||||||
|
|
||||||
|
if video_reader.get('frame_total') > 0:
|
||||||
|
video_frame_end = min(video_frame_end, video_reader.get('frame_total'))
|
||||||
|
|
||||||
with thread_semaphore():
|
with thread_semaphore():
|
||||||
conditional_set_video_frame_position(video_capture, video_frame_position)
|
conditional_set_video_reader_position(video_reader, video_frame_position)
|
||||||
|
|
||||||
for frame_number in range(video_frame_position, min(video_frame_position + chunk_size, video_frame_total)):
|
for frame_number in range(video_frame_position, video_frame_end):
|
||||||
has_vision_frame, vision_frame = video_capture.read()
|
has_vision_frame, vision_frame = read_video_reader_frame(video_reader)
|
||||||
|
|
||||||
if has_vision_frame:
|
if has_vision_frame:
|
||||||
video_frame_chunk[frame_number] = vision_frame
|
video_frame_chunk[frame_number] = vision_frame
|
||||||
|
|||||||
Reference in New Issue
Block a user