Skip to content

lombard_synth

Author: Bagus Tris Atmaja (with Claude Code) Affiliation: NAIST Date: 2026.09

Synthetic Lombard speech construction for the Lombard machine speech chain (Novitasari et al., IEEE/ACM TASLP 2022, Sec. IV-A). Following the paper, normal speech is turned into Lombard speech by modifying its pitch, intensity and duration with SoX, using the vocal changes observed in natural Lombard speech. One Lombard version is generated for each noise condition.

For a dumped SpeeChain dataset (idx2wav, idx2text, idx2duration, ...) this script produces, for every condition cond and subset:

{output_path}/{cond}/{subset}/idx2wav        Lombard (or clean) waveforms
{output_path}/{cond}/{subset}/idx2wav_len    number of samples
{output_path}/{cond}/{subset}/idx2wav_ref    the original clean waveform
{output_path}/{cond}/{subset}/idx2text       phoneme sequence (copied)
{output_path}/{cond}/{subset}/idx2duration   phoneme durations (rescaled by tempo)
{output_path}/{cond}/{subset}/idx2text_asr   raw transcript for the ASR feedback
{output_path}/{cond}/{subset}/idx2snr_cond   the condition name

Utterance indices are suffixed by -{cond} so that the conditions can be concatenated in one data_cfg. It also creates the noise waveforms used for the simulation ({output_path}/noise/white.wav and babble.wav, the latter being a sum of several utterances of a babble corpus, e.g. LibriSpeech dev-clean).

fix_stale_paths(idx2wav, base_dir, subset)

The dumped idx2wav files may hold absolute paths of another machine. If a path doesn't exist, it is relocated by (1) replacing the part before /datasets/ or /data/ by the data/ folder of the toolkit, (2) replacing the part before /{subset}/ by base_dir, or (3) looking for the file name directly under base_dir.

Source code in speechain/datasets/pyscripts/lombard_synth.py
def fix_stale_paths(
    idx2wav: Dict[str, str], base_dir: str, subset: str
) -> Dict[str, str]:
    """The dumped idx2wav files may hold absolute paths of another machine. If a path doesn't
    exist, it is relocated by (1) replacing the part before `/datasets/` or `/data/` by the `data/`
    folder of the toolkit, (2) replacing the part before `/{subset}/` by `base_dir`, or (3) looking
    for the file name directly under `base_dir`."""
    data_root = parse_path_args("data")
    fixed = {}
    for idx, path in idx2wav.items():
        if not os.path.exists(path):
            candidates = []
            for marker in ["/datasets/", "/data/"]:
                if marker in path:
                    candidates.append(os.path.join(data_root, path.split(marker, 1)[1]))
            if f"/{subset}/" in path:
                candidates.append(
                    os.path.join(base_dir, path.split(f"/{subset}/", 1)[1])
                )
            candidates.append(os.path.join(base_dir, os.path.basename(path)))
            for cand in candidates:
                if os.path.exists(cand):
                    path = cand
                    break
            else:
                raise FileNotFoundError(f"Cannot locate the waveform of {idx}: {path}")
        fixed[idx] = path
    return fixed

make_noise_files(args)

Generate white and babble noise waveforms.

Source code in speechain/datasets/pyscripts/lombard_synth.py
def make_noise_files(args):
    """Generate white and babble noise waveforms."""
    noise_dir = os.path.join(args.output_path, "noise")
    os.makedirs(noise_dir, exist_ok=True)
    rng = np.random.default_rng(args.seed)
    n_samples = int(args.noise_duration * args.sample_rate)

    white_path = os.path.join(noise_dir, "white.wav")
    if not os.path.exists(white_path):
        sf.write(
            white_path,
            (rng.standard_normal(n_samples) * 0.1).astype(np.float32),
            args.sample_rate,
        )

    babble_path = os.path.join(noise_dir, "babble.wav")
    if args.babble_idx2wav is not None and not os.path.exists(babble_path):
        import torch
        import torchaudio

        babble_file = parse_path_args(args.babble_idx2wav)
        idx2wav = fix_stale_paths(
            load_idx2data_file(babble_file),
            os.path.dirname(babble_file),
            os.path.basename(os.path.dirname(babble_file)),
        )
        paths = list(idx2wav.values())
        babble = np.zeros(n_samples, dtype=np.float32)
        for _ in range(args.babble_speakers):
            # one "speaker" stream = concatenation of random utterances
            stream, cursor = [], 0
            while cursor < n_samples:
                wav, sr = read_data_by_path(
                    paths[rng.integers(len(paths))], return_sample_rate=True
                )
                wav = wav.squeeze(-1).astype(np.float32)
                if sr != args.sample_rate:
                    wav = torchaudio.functional.resample(
                        torch.from_numpy(wav), sr, args.sample_rate
                    ).numpy()
                stream.append(wav)
                cursor += len(wav)
            stream = np.concatenate(stream)[:n_samples]
            babble += stream / (np.sqrt(np.mean(stream**2)) + 1e-8)
        babble = babble / np.max(np.abs(babble)) * 0.9
        sf.write(babble_path, babble, args.sample_rate)
    return noise_dir

sox_lombard(item, output_dir, rule)

Convert one clean waveform into its Lombard version by SoX.

Source code in speechain/datasets/pyscripts/lombard_synth.py
def sox_lombard(item, output_dir: str, rule: Dict) -> (str, str, int):
    """Convert one clean waveform into its Lombard version by SoX."""
    idx, src = item
    dst = os.path.join(output_dir, os.path.basename(src))
    if not os.path.exists(dst):
        effects = []
        if rule.get("pitch_cents", 0) != 0:
            effects += ["pitch", str(rule["pitch_cents"])]
        if rule.get("tempo", 1.0) != 1.0:
            effects += ["tempo", "-s", str(rule["tempo"])]
        if rule.get("gain_db", 0.0) != 0.0:
            # the limiter (-l) avoids clipping when the speech is amplified
            effects += ["gain", "-l", str(rule["gain_db"])]
        subprocess.run(
            ["sox", src, dst] + effects, check=True, stderr=subprocess.DEVNULL
        )
    info = sf.info(dst)
    return idx, dst, info.frames