mirror of
https://github.com/facefusion/facefusion.git
synced 2026-08-08 10:08:37 +02:00
* mark as next * unify the dependency checks in pre_check and add ffprobe (#1181) * drop keep_temp and the common options component (#1180) Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> * introduce ffprobe and ffprobe_builder (#1182) * introduce ffprobe and ffprobe_builder Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * introduce ffprobe and ffprobe_builder Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> * probe video metadata via ffprobe in vision (#1184) * probe video metadata via ffprobe in vision Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * probe video metadata via ffprobe in vision Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> * adopt the workflow task vocabulary from next major (#1185) Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> * introduce workflow-mode and workflow-strategy like next major (#1187) Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> * restrict hdr color transfer and tag the merge output as bt709 (#1188) * restrict hdr color transfer and tag the merge output as bt709 Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * restrict hdr color transfer and tag the merge output as bt709 Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * full video migration * compose the hdr fixture via the builder chain Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * compose the test fixtures via the builder and run_ffmpeg Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> * compose every test fixture via the builder and run_ffmpeg (#1189) * compose every test fixture via the builder and run_ffmpeg Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * use loops in tests for ffmpeg stuff --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> * New Video Manager (#1191) * tiny adjustment for tests * address the review on the video manager Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * introduce the stream strategy for the video workflow (#1192) * introduce the stream strategy for the video workflow Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * address the review on the stream strategy Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> * annotate the changes for review Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * annotate the new tests for review Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * match the temp pixel format help to the locale style Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * question the set_input_seek naming Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * question the reader and writer keys Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * drop the review annotations from the encoder mapping tests Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * drop the review annotations from the thread count tests Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * drop the review annotations from the ui files Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * capture the open review questions as annotations Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a * drop the settled annotations from the types Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Tbcd6VWCiU4BQP1gywPr2a --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> * switch to ffmpeg.style for audio.py * remove todos that were never needed * fix for ffmpeg7 * Add frame_store module (#1194) * add frame_store module * rename and change tests * rename and update tests * route window read through frame_store (#1196) * route window read through frame_store * update proper id * restore todos * restore todos * go v4 style for workflow (#1197) * go v4 style for workflow * remove some todos * route chunk read through frame_store (#1198) * vision integration * Deleted read_video_chunk + read_static_video_chunk * margin decouple (#1199) * fix windows CI fail (#1200) * Cleanup Part1 (#1201) * remove some todos, improve video manager, simplify ffmpeg commands and more * do more * remove thread count for filters * Cleanup Part 2 (#1202) * tons of renaming * tons of renaming * multi reader approach * bring tests to an okay-ish state * bring drain back * improve read_video_frame speed * rename method * move variables * seek video reader only when trim frame start is larger 0 * make stream the default * Cleanup/part 3 (#1203) * remove todo * sort out workflow, to match upcoming v4 * remove look ahead * remove core namespace again * Revamp execution provider overrides/adjustments (#1206) * Split provider hooks into override/adjust with cached CoreML base Replace the single resolve_inference_providers processor hook with two: override_inference_providers (full replacement) and adjust_inference_providers (merge options onto the base providers built by create_inference_providers). This lets CoreML processors inherit ModelCacheDirectory + SpecializationStrategy from the base while layering ModelFormat/MLComputeUnits on top. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01HTQCZiYjJyUX11bDpbRSiB * fix caching for execution provider by having override and adjust ways * fix caching for execution provider by having override and adjust ways * fix lint * use proper pytest fixtures --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix update preview bug (#1205) * fix update preview bug * fix update preview bug * remove guard * add is_vision_frame * Restrict the preview frame slider and the reader seek to the last frame index (#1207) * fix index bug * fix rounding bug * avoid tobytes copy (#1208) * beautify tests * hide ffmpeg warnings * simplify process_stream_frame * Use is vision frame everywhere (#1210) * use is_vision_frame everywhere * fix hash * fix lint * fix hash creation in face store * that model does not exist * update workflow ffmpeg * guard workflow (#1211) * bump version and dependencies * Update preview * switch workflow strategy to disk|memory * update preview * update preview * fix wording * last minute change workflow position * adjust wording --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com> Co-authored-by: Harisreedhar <46858047+harisreedhar@users.noreply.github.com> Co-authored-by: harisreedhar <h4harisreedhar.s.s@gmail.com>
144 lines
4.7 KiB
Python
144 lines
4.7 KiB
Python
from functools import lru_cache
|
|
from typing import Any, List, Optional
|
|
|
|
import numpy
|
|
import scipy
|
|
from numpy.typing import NDArray
|
|
|
|
from facefusion import ffmpeg
|
|
from facefusion.filesystem import is_audio
|
|
from facefusion.types import Audio, AudioFrame, Fps, Mel, MelFilterBank, Spectrogram
|
|
from facefusion.voice_extractor import batch_extract_voice
|
|
|
|
|
|
@lru_cache(maxsize = 64)
|
|
def read_static_audio(audio_path : str, fps : Fps) -> Optional[List[AudioFrame]]:
|
|
return read_audio(audio_path, fps)
|
|
|
|
|
|
def read_audio(audio_path : str, fps : Fps) -> Optional[List[AudioFrame]]:
|
|
audio_sample_rate = 48000
|
|
audio_sample_size = 16
|
|
audio_channel_total = 2
|
|
|
|
if is_audio(audio_path):
|
|
audio_buffer = ffmpeg.read_audio_buffer(audio_path, audio_sample_rate, audio_sample_size, audio_channel_total)
|
|
audio = numpy.frombuffer(audio_buffer, dtype = numpy.int16).reshape(-1, 2)
|
|
audio = prepare_audio(audio)
|
|
spectrogram = create_spectrogram(audio)
|
|
audio_frames = extract_audio_frames(spectrogram, fps)
|
|
return audio_frames
|
|
return None
|
|
|
|
|
|
@lru_cache(maxsize = 64)
|
|
def read_static_voice(audio_path : str, fps : Fps) -> Optional[List[AudioFrame]]:
|
|
return read_voice(audio_path, fps)
|
|
|
|
|
|
def read_voice(audio_path : str, fps : Fps) -> Optional[List[AudioFrame]]:
|
|
voice_sample_rate = 48000
|
|
voice_sample_size = 16
|
|
voice_channel_total = 2
|
|
voice_chunk_size = 240 * 1024
|
|
voice_step_size = 180 * 1024
|
|
|
|
if is_audio(audio_path):
|
|
audio_buffer = ffmpeg.read_audio_buffer(audio_path, voice_sample_rate, voice_sample_size, voice_channel_total)
|
|
audio = numpy.frombuffer(audio_buffer, dtype = numpy.int16).reshape(-1, 2)
|
|
audio = batch_extract_voice(audio, voice_chunk_size, voice_step_size)
|
|
audio = prepare_voice(audio)
|
|
spectrogram = create_spectrogram(audio)
|
|
audio_frames = extract_audio_frames(spectrogram, fps)
|
|
return audio_frames
|
|
return None
|
|
|
|
|
|
def get_audio_frame(audio_path : str, fps : Fps, frame_number : int = 0) -> Optional[AudioFrame]:
|
|
if is_audio(audio_path):
|
|
audio_frames = read_static_audio(audio_path, fps)
|
|
if frame_number in range(len(audio_frames)):
|
|
return audio_frames[frame_number]
|
|
return None
|
|
|
|
|
|
def extract_audio_frames(spectrogram : Spectrogram, fps : Fps) -> List[AudioFrame]:
|
|
audio_frames = []
|
|
mel_filter_total = 80
|
|
audio_step_size = 16
|
|
indices = numpy.arange(0, spectrogram.shape[1], mel_filter_total / fps).astype(numpy.int16)
|
|
indices = indices[indices >= audio_step_size]
|
|
|
|
for index in indices:
|
|
start = max(0, index - audio_step_size)
|
|
audio_frames.append(spectrogram[:, start:index])
|
|
|
|
return audio_frames
|
|
|
|
|
|
def get_voice_frame(audio_path : str, fps : Fps, frame_number : int = 0) -> Optional[AudioFrame]:
|
|
if is_audio(audio_path):
|
|
voice_frames = read_static_voice(audio_path, fps)
|
|
if frame_number in range(len(voice_frames)):
|
|
return voice_frames[frame_number]
|
|
return None
|
|
|
|
|
|
def create_empty_audio_frame() -> AudioFrame:
|
|
mel_filter_total = 80
|
|
audio_step_size = 16
|
|
audio_frame = numpy.zeros((mel_filter_total, audio_step_size)).astype(numpy.int16)
|
|
return audio_frame
|
|
|
|
|
|
def prepare_audio(audio : Audio) -> Audio:
|
|
if audio.ndim > 1:
|
|
audio = numpy.mean(audio, axis = 1)
|
|
audio = audio / numpy.max(numpy.abs(audio), axis = 0)
|
|
audio = scipy.signal.lfilter([ 1.0, -0.97 ], [ 1.0 ], audio)
|
|
return audio
|
|
|
|
|
|
def prepare_voice(audio : Audio) -> Audio:
|
|
audio_sample_rate = 48000
|
|
audio_resample_rate = 16000
|
|
audio_resample_factor = round(len(audio) * audio_resample_rate / audio_sample_rate)
|
|
audio = scipy.signal.resample(audio, audio_resample_factor)
|
|
audio = prepare_audio(audio)
|
|
return audio
|
|
|
|
|
|
def convert_hertz_to_mel(hertz : float) -> float:
|
|
return 2595 * numpy.log10(1 + hertz / 700)
|
|
|
|
|
|
def convert_mel_to_hertz(mel : Mel) -> NDArray[Any]:
|
|
return 700 * (10 ** (mel / 2595) - 1)
|
|
|
|
|
|
def create_mel_filter_bank() -> MelFilterBank:
|
|
audio_sample_rate = 16000
|
|
audio_frequency_min = 55.0
|
|
audio_frequency_max = 7600.0
|
|
mel_filter_total = 80
|
|
mel_bin_total = 800
|
|
mel_filter_bank = numpy.zeros((mel_filter_total, mel_bin_total // 2 + 1))
|
|
mel_frequency_range = numpy.linspace(convert_hertz_to_mel(audio_frequency_min), convert_hertz_to_mel(audio_frequency_max), mel_filter_total + 2)
|
|
indices = numpy.floor((mel_bin_total + 1) * convert_mel_to_hertz(mel_frequency_range) / audio_sample_rate).astype(numpy.int16)
|
|
|
|
for index in range(mel_filter_total):
|
|
start = indices[index]
|
|
end = indices[index + 1]
|
|
mel_filter_bank[index, start:end] = scipy.signal.windows.triang(end - start)
|
|
|
|
return mel_filter_bank
|
|
|
|
|
|
def create_spectrogram(audio : Audio) -> Spectrogram:
|
|
mel_bin_total = 800
|
|
mel_bin_overlap = 600
|
|
mel_filter_bank = create_mel_filter_bank()
|
|
spectrogram = scipy.signal.stft(audio, nperseg = mel_bin_total, nfft = mel_bin_total, noverlap = mel_bin_overlap)[2]
|
|
spectrogram = numpy.dot(mel_filter_bank, numpy.abs(spectrogram))
|
|
return spectrogram
|