#!/bin/bash set -euo pipefail PROGPATH=$(realpath "$0") PROGDIR=$(dirname "${PROGPATH}") source "${PROGDIR}/shell.functions" if [ "${HELP}" == "true" ] ; then cat <&2 Usage: $0 [ --cleanup-archives ] [ --cleanup-input-files ] [ --data-dir= ] --cleanup-archives : Automatically clean up any downloaded archvies after extraction. --cleanup-intermediate-files : Automatically clean up the intermediate files after they've : converted to 16k. : Path to the data directory. : Default: ${DATA_DIR} EOF exit 1 fi mkdir -p "${DATA_DIR}/training_datasets/downloads" || : cd "${DATA_DIR}/training_datasets" HF_RIR_REPO_ID="TaterTotterson/MIT_environmental_impulse_responses" HF_RIR_API_URL="https://huggingface.co/api/datasets/${HF_RIR_REPO_ID}" HF_RIR_SOURCE_KEY="hf_mit_environmental_impulse_responses" AUDIO_DIR="./mit_rirs" mkdir -p "${AUDIO_DIR}" || : AUDIO16K_DIR="./mit_rirs_16k" mkdir -p "${AUDIO16K_DIR}" || : AUDIO_FILECOUNT="./downloads/mit_rir_filecount" AUDIO_IN_GLOB="*.wav" declare -A filecounts=( [${HF_RIR_SOURCE_KEY}]=0 ) get_filecounts filecounts "${AUDIO_FILECOUNT}" echo "===== Checking MIT environmental RIRs =====" download_hf_mit_rirs() { source ${DATA_DIR}/.venv/bin/activate python - "${HF_RIR_REPO_ID}" "${HF_RIR_API_URL}" "${AUDIO_DIR}" <<-'EOF' import json import sys import urllib.parse import urllib.request from pathlib import Path repo_id = sys.argv[1] api_url = sys.argv[2] audio_dir = Path(sys.argv[3]) audio_dir.mkdir(parents=True, exist_ok=True) request = urllib.request.Request(api_url, headers={"User-Agent": "WakeWordTrainer/1.0"}) with urllib.request.urlopen(request, timeout=30) as response: metadata = json.loads(response.read().decode("utf-8")) files = sorted( sibling.get("rfilename", "") for sibling in metadata.get("siblings", []) if str(sibling.get("rfilename", "")).startswith("16khz/") and str(sibling.get("rfilename", "")).lower().endswith(".wav") ) if not files: raise SystemExit("Hugging Face MIT RIR dataset did not list any 16khz WAV files") downloaded = 0 skipped = 0 for rel in files: target = audio_dir / rel if target.exists() and target.stat().st_size > 0: skipped += 1 continue target.parent.mkdir(parents=True, exist_ok=True) encoded = urllib.parse.quote(rel, safe="/") url = f"https://huggingface.co/datasets/{repo_id}/resolve/main/{encoded}" with urllib.request.urlopen(url, timeout=60) as response: target.write_bytes(response.read()) if not target.exists() or target.stat().st_size == 0: raise SystemExit(f"download failed for {rel}") downloaded += 1 print(f" Hugging Face MIT environmental RIR download complete ({downloaded} downloaded, {skipped} reused)") print(f" MIT environmental RIR files available: {len(files)}") EOF } converter() { source ${DATA_DIR}/.venv/bin/activate python - "${AUDIO_DIR}" "${AUDIO16K_DIR}" <<-EOF import os, sys, subprocess, scipy.io.wavfile, numpy as np from pathlib import Path import soundfile as sf import librosa from tqdm import tqdm def write_wav(dst: Path, data: np.ndarray, sr: int): x = np.clip(data, -1.0, 1.0) scipy.io.wavfile.write(dst, sr, (x * 32767).astype(np.int16)) rir_in = Path(sys.argv[1]) rir_out = Path(sys.argv[2]) waves = list(rir_in.rglob("*.wav")) try: print(" MIT environmental RIR normalizing to 16k…") # Normalize to 16k mono for p in tqdm(waves, desc=" MIT environmental RIR (resample 16k mono)"): outfile = Path(rir_out / p.name) if outfile.exists(): continue a, sr = sf.read(p, always_2d=False) if a.ndim > 1: a = a[:, 0] if sr != 16000: a, _ = librosa.load(p, sr=16000, mono=True) write_wav(outfile, a, 16000) print(" MIT environmental RIR normalization complete") except Exception as e2: print(f" MIT environmental RIR preparation failed: {e2}") raise EOF } expected_filecount=${filecounts[${HF_RIR_SOURCE_KEY}]} actual_filecount=$(find "${AUDIO16K_DIR}" -name '*.wav' 2>/dev/null | wc -l) || : write_filecount=false if [ "${actual_filecount}" -ne 0 ] && [ "${actual_filecount}" -eq "${expected_filecount}" ] ; then echo " Existing ${AUDIO16K_DIR} valid" else actual_filecount=$(find "${AUDIO_DIR}" -name "${AUDIO_IN_GLOB}" 2>/dev/null | wc -l) || : if [ "${actual_filecount}" -eq 0 ] || [ "${expected_filecount}" -eq 0 ] || [ "${actual_filecount}" -ne "${expected_filecount}" ] ; then rm -rf "${AUDIO_DIR}" || : mkdir -p "${AUDIO_DIR}" || : echo " Downloading MIT environmental impulse responses from Hugging Face mirror" download_hf_mit_rirs fi converter actual_filecount=$(find "${AUDIO16K_DIR}" -name "*.wav" 2>/dev/null | wc -l) || : filecounts[${HF_RIR_SOURCE_KEY}]="${actual_filecount}" write_filecount=true fi if ${write_filecount} ; then write_filecounts filecounts "${AUDIO_FILECOUNT}" fi if "${CLEANUP_INTERMEDIATE_FILES}" && [ -d "${AUDIO_DIR}" ]; then echo " Cleaning up ${AUDIO_DIR}" rm -rf "${AUDIO_DIR}" fi echo " MIT environmental RIRs complete" exit 0