| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146 |
- import os
- import sys
- import time
- import socket
- import struct
- import numpy as np
- # Network settings
- UDP_IP = os.environ.get("UDP_IP", "239.0.0.1")
- UDP_PORT = int(os.environ.get("UDP_PORT", "11988"))
- # Audio stream constants
- SAMPLE_RATE = 44100
- CHUNK_SIZE = 1024
- CHUNK_DURATION = CHUNK_SIZE / SAMPLE_RATE
- FRAME_BYTES = CHUNK_SIZE * 4
- MAX_SILENT_FRAMES = 43 # ~1 second of silence at 43.06 FPS
- # DSP tuning environment overrides
- FREQ_MIN = float(os.environ.get("FREQ_MIN", "40.0")) # Catch deep sub-bass and 50Hz kicks
- FREQ_MAX = float(os.environ.get("FREQ_MAX", "12000.0")) # Capture crisp cymbals and high transients
- SILENCE_THRESHOLD = float(os.environ.get("SILENCE_THRESHOLD", "0.5"))
- # Dynamic AGC and Frequency Tilt overrides
- GAIN_MIN = float(os.environ.get("GAIN_MIN", "800.0")) # Floor: prevents squashing heavily mastered EDM
- GAIN_MAX = float(os.environ.get("GAIN_MAX", "8500.0")) # Ceiling: prevents boosting background tape hiss
- TILT_EXPONENT = float(os.environ.get("TILT_EXPONENT", "0.42")) # Logarithmic treble compensation curve
- DECAY_RATE = float(os.environ.get("DECAY_RATE", "0.998")) # ~10-second slow recovery decay per frame
- sock = socket.socket(socket.AF_INET, socket.SOCK_DGRAM, socket.IPPROTO_UDP)
- sock.setsockopt(socket.IPPROTO_IP, socket.IP_MULTICAST_TTL, 2)
- # Pre-compute Hanning window to save CPU cycles inside the loop
- HANNING_WINDOW = np.hanning(CHUNK_SIZE).astype(np.float32)
- # Compute 16 logarithmic frequency bands
- FREQ_EDGES = np.logspace(np.log10(FREQ_MIN), np.log10(FREQ_MAX), 17)
- fft_freqs = np.fft.rfftfreq(CHUNK_SIZE, 1.0 / SAMPLE_RATE)
- bins_idx = []
- for i in range(16):
- low = FREQ_EDGES[i]
- high = FREQ_EDGES[i + 1]
- idx = np.where((fft_freqs >= low) & (fft_freqs < high) & (fft_freqs > 0))[0]
- if len(idx) == 0:
- non_zero_bins = np.where(fft_freqs > 0)[0]
- closest = non_zero_bins[np.argmin(np.abs(fft_freqs[non_zero_bins] - low))]
- idx = [closest]
- bins_idx.append(idx)
- # Pre-compute logarithmic treble tilt weights (1/f pink noise balance)
- band_centers = np.sqrt(FREQ_EDGES[:-1] * FREQ_EDGES[1:])
- TILT_WEIGHTS = ((band_centers / FREQ_MIN) ** TILT_EXPONENT).astype(np.float32)
- # State variables
- sample_smth = 0.0
- silence_frames = 0
- running_peak = 0.05 # Initial baseline ceiling
- STRUCT_FMT_V2 = "<6s2xffB3x16sd"
- start_time = None
- frames_processed = 0
- while True:
- t0 = time.perf_counter()
- raw_data = sys.stdin.buffer.read(FRAME_BYTES)
- read_duration = time.perf_counter() - t0
- if not raw_data or len(raw_data) < FRAME_BYTES:
- break
- # Gap Detector: If the pipe sat empty for >200ms, reset the clock & AGC baseline
- if read_duration > 0.2:
- start_time = time.perf_counter()
- frames_processed = 0
- running_peak = 0.05
- if start_time is None:
- start_time = time.perf_counter()
- # Audio ingestion and magnitude extraction
- audio = np.frombuffer(raw_data, dtype=np.int16).astype(np.float32)
- left = audio[0::2]
- right = audio[1::2]
- mono = (left + right) * (1.0 / 65536.0)
- raw_mag = float(np.max(np.abs(mono)) * 255.0)
- sample_smth = 0.7 * sample_smth + 0.3 * raw_mag
- sample_peak = 1 if raw_mag > (sample_smth * 1.5 + 20) else 0
- if raw_mag < SILENCE_THRESHOLD:
- silence_frames += 1
- else:
- silence_frames = 0
- if silence_frames <= MAX_SILENT_FRAMES:
- windowed = mono * HANNING_WINDOW
- fft_vals = np.abs(np.fft.rfft(windowed)) * (2.0 / CHUNK_SIZE)
- # Vectorized band extraction
- raw_energies = np.empty(16, dtype=np.float32)
- for i in range(16):
- raw_energies[i] = np.mean(fft_vals[bins_idx[i]])
- # Apply logarithmic pink noise compensation
- tilted_energies = raw_energies * TILT_WEIGHTS
- # Asymmetric AGC: Instant attack, slow crawl decay
- current_max = float(np.max(tilted_energies))
- if current_max > running_peak:
- running_peak = current_max
- else:
- running_peak = max(0.005, running_peak * DECAY_RATE)
- # Dynamic gain bounded within strict sanity limits
- dynamic_gain = np.clip(255.0 / running_peak, GAIN_MIN, GAIN_MAX)
- # Scale into 8-bit unsigned integer array
- scaled = np.clip(tilted_energies * dynamic_gain, 0, 255).astype(np.uint8)
- fft_result = bytes(scaled)
- payload = struct.pack(
- STRUCT_FMT_V2,
- b"00002\x00",
- float(raw_mag),
- float(sample_smth),
- sample_peak,
- fft_result,
- float(raw_mag)
- )
- try:
- sock.sendto(payload, (UDP_IP, UDP_PORT))
- except Exception:
- pass
- # Metronome pacing
- frames_processed += 1
- target_time = start_time + (frames_processed * CHUNK_DURATION)
- sleep_time = target_time - time.perf_counter()
- if sleep_time > 0:
- time.sleep(sleep_time)
- elif sleep_time < -1.0:
- start_time = time.perf_counter()
- frames_processed = 0
|