Author: Bagus Tris Atmaja (with Claude Code)
Affiliation: NAIST
Date: 2026.09
Synthetic Lombard speech construction for the Lombard machine speech chain
(Novitasari et al., IEEE/ACM TASLP 2022, Sec. IV-A). Following the paper,
normal speech is turned into Lombard speech by modifying its pitch,
intensity and duration with SoX, using the vocal changes observed in natural
Lombard speech. One Lombard version is generated for each noise condition.
For a dumped SpeeChain dataset (idx2wav, idx2text, idx2duration, ...) this
script produces, for every condition cond and subset:
{output_path}/{cond}/{subset}/idx2wav Lombard (or clean) waveforms
{output_path}/{cond}/{subset}/idx2wav_len number of samples
{output_path}/{cond}/{subset}/idx2wav_ref the original clean waveform
{output_path}/{cond}/{subset}/idx2text phoneme sequence (copied)
{output_path}/{cond}/{subset}/idx2duration phoneme durations (rescaled by tempo)
{output_path}/{cond}/{subset}/idx2text_asr raw transcript for the ASR feedback
{output_path}/{cond}/{subset}/idx2snr_cond the condition name
Utterance indices are suffixed by -{cond} so that the conditions can be
concatenated in one data_cfg. It also creates the noise waveforms used for
the simulation ({output_path}/noise/white.wav and babble.wav, the latter
being a sum of several utterances of a babble corpus, e.g. LibriSpeech
dev-clean).
fix_stale_paths(idx2wav, base_dir, subset)
The dumped idx2wav files may hold absolute paths of another machine. If a path doesn't
exist, it is relocated by (1) replacing the part before /datasets/ or /data/ by the data/
folder of the toolkit, (2) replacing the part before /{subset}/ by base_dir, or (3) looking
for the file name directly under base_dir.
Source code in speechain/datasets/pyscripts/lombard_synth.py
| def fix_stale_paths(
idx2wav: Dict[str, str], base_dir: str, subset: str
) -> Dict[str, str]:
"""The dumped idx2wav files may hold absolute paths of another machine. If a path doesn't
exist, it is relocated by (1) replacing the part before `/datasets/` or `/data/` by the `data/`
folder of the toolkit, (2) replacing the part before `/{subset}/` by `base_dir`, or (3) looking
for the file name directly under `base_dir`."""
data_root = parse_path_args("data")
fixed = {}
for idx, path in idx2wav.items():
if not os.path.exists(path):
candidates = []
for marker in ["/datasets/", "/data/"]:
if marker in path:
candidates.append(os.path.join(data_root, path.split(marker, 1)[1]))
if f"/{subset}/" in path:
candidates.append(
os.path.join(base_dir, path.split(f"/{subset}/", 1)[1])
)
candidates.append(os.path.join(base_dir, os.path.basename(path)))
for cand in candidates:
if os.path.exists(cand):
path = cand
break
else:
raise FileNotFoundError(f"Cannot locate the waveform of {idx}: {path}")
fixed[idx] = path
return fixed
|
make_noise_files(args)
Generate white and babble noise waveforms.
Source code in speechain/datasets/pyscripts/lombard_synth.py
| def make_noise_files(args):
"""Generate white and babble noise waveforms."""
noise_dir = os.path.join(args.output_path, "noise")
os.makedirs(noise_dir, exist_ok=True)
rng = np.random.default_rng(args.seed)
n_samples = int(args.noise_duration * args.sample_rate)
white_path = os.path.join(noise_dir, "white.wav")
if not os.path.exists(white_path):
sf.write(
white_path,
(rng.standard_normal(n_samples) * 0.1).astype(np.float32),
args.sample_rate,
)
babble_path = os.path.join(noise_dir, "babble.wav")
if args.babble_idx2wav is not None and not os.path.exists(babble_path):
import torch
import torchaudio
babble_file = parse_path_args(args.babble_idx2wav)
idx2wav = fix_stale_paths(
load_idx2data_file(babble_file),
os.path.dirname(babble_file),
os.path.basename(os.path.dirname(babble_file)),
)
paths = list(idx2wav.values())
babble = np.zeros(n_samples, dtype=np.float32)
for _ in range(args.babble_speakers):
# one "speaker" stream = concatenation of random utterances
stream, cursor = [], 0
while cursor < n_samples:
wav, sr = read_data_by_path(
paths[rng.integers(len(paths))], return_sample_rate=True
)
wav = wav.squeeze(-1).astype(np.float32)
if sr != args.sample_rate:
wav = torchaudio.functional.resample(
torch.from_numpy(wav), sr, args.sample_rate
).numpy()
stream.append(wav)
cursor += len(wav)
stream = np.concatenate(stream)[:n_samples]
babble += stream / (np.sqrt(np.mean(stream**2)) + 1e-8)
babble = babble / np.max(np.abs(babble)) * 0.9
sf.write(babble_path, babble, args.sample_rate)
return noise_dir
|
sox_lombard(item, output_dir, rule)
Convert one clean waveform into its Lombard version by SoX.
Source code in speechain/datasets/pyscripts/lombard_synth.py
| def sox_lombard(item, output_dir: str, rule: Dict) -> (str, str, int):
"""Convert one clean waveform into its Lombard version by SoX."""
idx, src = item
dst = os.path.join(output_dir, os.path.basename(src))
if not os.path.exists(dst):
effects = []
if rule.get("pitch_cents", 0) != 0:
effects += ["pitch", str(rule["pitch_cents"])]
if rule.get("tempo", 1.0) != 1.0:
effects += ["tempo", "-s", str(rule["tempo"])]
if rule.get("gain_db", 0.0) != 0.0:
# the limiter (-l) avoids clipping when the speech is amplified
effects += ["gain", "-l", str(rule["gain_db"])]
subprocess.run(
["sox", src, dst] + effects, check=True, stderr=subprocess.DEVNULL
)
info = sf.info(dst)
return idx, dst, info.frames
|