183 lines
7.7 KiB
Python
183 lines
7.7 KiB
Python
"""
|
|
Background Music Generation & Audio Synthesis Module.
|
|
Generates dynamic, mood-matched cinematic background music scores (ambient pads,
|
|
harmonic minor chord progressions, subtle sub-bass, atmospheric texture) for the exact
|
|
video length, perfectly structured for dynamic audio ducking under speech.
|
|
"""
|
|
|
|
import os
|
|
import math
|
|
import wave
|
|
import struct
|
|
import random
|
|
import logging
|
|
import numpy as np
|
|
from typing import Optional
|
|
|
|
logger = logging.getLogger("YTFactory.Music")
|
|
|
|
# Musical note frequencies (Hz)
|
|
NOTES = {
|
|
"C2": 65.41, "D2": 73.42, "E2": 82.41, "F2": 87.31, "G2": 98.00, "A2": 110.00, "B2": 123.47,
|
|
"C3": 130.81, "D3": 146.83, "E3": 164.81, "F3": 174.61, "G3": 196.00, "A3": 220.00, "B3": 246.94,
|
|
"C4": 261.63, "D4": 293.66, "E4": 329.63, "F4": 349.23, "G4": 392.00, "A4": 440.00, "B4": 493.88,
|
|
"C5": 523.25, "D5": 587.33, "E5": 659.25, "F5": 698.46, "G5": 783.99, "A5": 880.00, "B5": 987.77,
|
|
}
|
|
|
|
# Cinematic minor chord progressions (Root, Minor 3rd, 5th, Octave)
|
|
PROGRESSIONS = {
|
|
"d_minor": [
|
|
[NOTES["D3"], NOTES["F3"], NOTES["A3"], NOTES["D4"]],
|
|
[NOTES["B2"], NOTES["D3"], NOTES["F3"], NOTES["B3"]],
|
|
[NOTES["G2"], NOTES["B2"], NOTES["D3"], NOTES["G3"]],
|
|
[NOTES["A2"], NOTES["C3"], NOTES["E3"], NOTES["A3"]],
|
|
],
|
|
"a_minor": [
|
|
[NOTES["A2"], NOTES["C3"], NOTES["E3"], NOTES["A3"]],
|
|
[NOTES["F2"], NOTES["A2"], NOTES["C3"], NOTES["F3"]],
|
|
[NOTES["D2"], NOTES["F2"], NOTES["A2"], NOTES["D3"]],
|
|
[NOTES["E2"], NOTES["G2"], NOTES["B2"], NOTES["E3"]],
|
|
],
|
|
"c_minor": [
|
|
[NOTES["C3"], NOTES["D3"] * 1.06, NOTES["G3"], NOTES["C4"]],
|
|
[NOTES["A2"] * 1.06, NOTES["C3"], NOTES["E3"] * 1.06, NOTES["A3"] * 1.06],
|
|
[NOTES["F2"], NOTES["A2"] * 1.06, NOTES["C3"], NOTES["F3"]],
|
|
[NOTES["G2"], NOTES["B2"], NOTES["D3"], NOTES["G3"]],
|
|
]
|
|
}
|
|
|
|
class MusicGen:
|
|
def __init__(self, sample_rate: int = 44100):
|
|
self.sample_rate = sample_rate
|
|
|
|
def _generate_pad_chord(self, freqs: list, duration_sec: float) -> np.ndarray:
|
|
"""Synthesize a lush, detuned ambient pad chord with subtle LFO filtering."""
|
|
num_samples = int(self.sample_rate * duration_sec)
|
|
t = np.linspace(0, duration_sec, num_samples, endpoint=False)
|
|
chord_signal = np.zeros(num_samples, dtype=np.float32)
|
|
|
|
# LFO for slow ambient movement (0.2 Hz)
|
|
lfo = 0.5 + 0.5 * np.sin(2 * np.pi * 0.2 * t)
|
|
|
|
for freq in freqs:
|
|
# 3 detuned oscillators per note for stereo-like warmth
|
|
osc1 = np.sin(2 * np.pi * freq * t)
|
|
osc2 = 0.5 * np.sin(2 * np.pi * (freq * 1.003) * t)
|
|
osc3 = 0.5 * np.sin(2 * np.pi * (freq * 0.997) * t)
|
|
# Gentle soft harmonic
|
|
osc4 = 0.15 * np.sin(2 * np.pi * (freq * 2.0) * t)
|
|
|
|
note_signal = (osc1 + osc2 + osc3 + osc4) / 2.15
|
|
chord_signal += note_signal
|
|
|
|
# Normalize chord
|
|
chord_signal = (chord_signal / max(1, len(freqs))) * lfo
|
|
|
|
# Envelope: 1.5s attack, 1.5s release
|
|
attack_samples = int(self.sample_rate * min(1.5, duration_sec * 0.25))
|
|
release_samples = int(self.sample_rate * min(1.5, duration_sec * 0.25))
|
|
|
|
env = np.ones(num_samples, dtype=np.float32)
|
|
if attack_samples > 0:
|
|
env[:attack_samples] = np.linspace(0, 1, attack_samples)
|
|
if release_samples > 0:
|
|
env[-release_samples:] = np.linspace(1, 0, release_samples)
|
|
|
|
return chord_signal * env
|
|
|
|
def _generate_sub_bass(self, root_freq: float, duration_sec: float) -> np.ndarray:
|
|
"""Warm deep sub-bass oscillator (sine + soft 2nd harmonic)."""
|
|
num_samples = int(self.sample_rate * duration_sec)
|
|
t = np.linspace(0, duration_sec, num_samples, endpoint=False)
|
|
sub = np.sin(2 * np.pi * (root_freq / 2.0) * t)
|
|
sub_2 = 0.25 * np.sin(2 * np.pi * root_freq * t)
|
|
return (sub + sub_2) * 0.4
|
|
|
|
def _generate_texture(self, duration_sec: float) -> np.ndarray:
|
|
"""Gentle tape/atmosphere background texture."""
|
|
num_samples = int(self.sample_rate * duration_sec)
|
|
# Pinkish filtered noise
|
|
noise = np.random.normal(0, 0.02, num_samples).astype(np.float32)
|
|
# Simple moving average lowpass filter
|
|
kernel_size = 50
|
|
kernel = np.ones(kernel_size) / kernel_size
|
|
filtered_noise = np.convolve(noise, kernel, mode='same')
|
|
return filtered_noise
|
|
|
|
def generate(self, duration_sec: float, output_wav: str, mood: str = "cinematic") -> str:
|
|
"""
|
|
Synthesize a complete cinematic background music track tailored to the exact duration.
|
|
"""
|
|
logger.info(f"[MUSIC] Synthesizing {duration_sec:.1f}s cinematic soundtrack ({mood})...")
|
|
os.makedirs(os.path.dirname(os.path.abspath(output_wav)), exist_ok=True)
|
|
|
|
chord_prog = PROGRESSIONS["d_minor"] if mood != "space" else PROGRESSIONS["a_minor"]
|
|
chord_duration = 6.0 # 6 seconds per chord change
|
|
|
|
total_samples = int(self.sample_rate * duration_sec)
|
|
master_audio_left = np.zeros(total_samples, dtype=np.float32)
|
|
master_audio_right = np.zeros(total_samples, dtype=np.float32)
|
|
|
|
# Generate sequential chords
|
|
cur_sample = 0
|
|
chord_idx = 0
|
|
while cur_sample < total_samples:
|
|
remaining_samples = total_samples - cur_sample
|
|
this_dur = min(chord_duration, remaining_samples / self.sample_rate)
|
|
if this_dur <= 0.1:
|
|
break
|
|
|
|
chord = chord_prog[chord_idx % len(chord_prog)]
|
|
pad = self._generate_pad_chord(chord, this_dur)
|
|
bass = self._generate_sub_bass(chord[0], this_dur)
|
|
|
|
seg_samples = len(pad)
|
|
|
|
# Stereo spread: subtle phase offset
|
|
master_audio_left[cur_sample:cur_sample + seg_samples] += (pad * 0.7 + bass * 0.5)
|
|
master_audio_right[cur_sample:cur_sample + seg_samples] += (pad * 0.72 + bass * 0.48)
|
|
|
|
cur_sample += seg_samples
|
|
chord_idx += 1
|
|
|
|
# Add atmospheric texture layer
|
|
texture = self._generate_texture(duration_sec)
|
|
master_audio_left += texture[:total_samples]
|
|
master_audio_right += texture[:total_samples]
|
|
|
|
# Master Fade-in and Fade-out
|
|
fade_in_samples = int(self.sample_rate * 2.0)
|
|
fade_out_samples = int(self.sample_rate * 3.0)
|
|
|
|
if fade_in_samples < total_samples:
|
|
fade_in = np.linspace(0, 1, fade_in_samples)
|
|
master_audio_left[:fade_in_samples] *= fade_in
|
|
master_audio_right[:fade_in_samples] *= fade_in
|
|
|
|
if fade_out_samples < total_samples:
|
|
fade_out = np.linspace(1, 0, fade_out_samples)
|
|
master_audio_left[-fade_out_samples:] *= fade_out
|
|
master_audio_right[-fade_out_samples:] *= fade_out
|
|
|
|
# Normalize to -18 dB peak so it is ready for background ducking
|
|
peak = max(np.max(np.abs(master_audio_left)), np.max(np.abs(master_audio_right)), 1e-6)
|
|
target_peak = 0.35 # ~ -18 dBFS
|
|
master_audio_left = (master_audio_left / peak) * target_peak
|
|
master_audio_right = (master_audio_right / peak) * target_peak
|
|
|
|
# Convert to 16-bit PCM stereo
|
|
left_int16 = (master_audio_left * 32767).astype(np.int16)
|
|
right_int16 = (master_audio_right * 32767).astype(np.int16)
|
|
|
|
stereo = np.empty((total_samples, 2), dtype=np.int16)
|
|
stereo[:, 0] = left_int16
|
|
stereo[:, 1] = right_int16
|
|
|
|
with wave.open(output_wav, 'wb') as wf:
|
|
wf.setnchannels(2)
|
|
wf.setsampwidth(2)
|
|
wf.setframerate(self.sample_rate)
|
|
wf.writeframes(stereo.tobytes())
|
|
|
|
logger.info(f"[MUSIC] Soundtrack generated successfully: {output_wav}")
|
|
return output_wav
|