MultiTalkFT
Fine-tuning corpus for full-duplex multi-speaker dialogue.
Schemas
data_{zh,en}{,_multichannel}.jsonl (one record per line):
| field | type | description |
|---|---|---|
path | string | relative path to the audio file |
voice | string | relative path to speaker prompt |
duration | float | clip duration in seconds |
system | string | persona / system prompt |
transcripts/*.parquet:
| column | type | description |
|---|---|---|
audio_path | string | matches data_*.jsonl path |
id | string | |
duration | float | |
num_channels | int32 | original conversation speaker count |
speaker_to_channel | string | JSON-encoded {speaker: channel_index} |
voice | string | JSON-encoded {speaker: relative voice path} |
alignments | string | JSON-encoded flat list [[word, [start, end], speaker_label], …] |
training | string | JSON-encoded {system_prompt, voice_prompt (relative), …} |
Quick load
from datasets import load_dataset
from huggingface_hub import hf_hub_download
import json, soundfile as sf
REPO = "MultiTalk/MultiTalkFT"
# 1) 100-row sample preview.
preview = load_dataset(REPO, "preview", split="preview")
print(preview[0]) # {audio: <rel_path>, duration, lang, alignments}
# 2) Full manifests — pull jsonl files directly.
for name in ("data_zh.jsonl", "data_en.jsonl",
"data_zh_multichannel.jsonl", "data_en_multichannel.jsonl"):
p = hf_hub_download(REPO, name, repo_type="dataset")
print(name, sum(1 for _ in open(p)), "rows")
# 3) Word-level transcripts (sharded parquet).
ts = load_dataset(
"parquet",
data_files=f"https://huggingface.co/datasets/{REPO}/resolve/main/transcripts/zh-*.parquet",
split="train", streaming=True,
)
for rec in ts.take(1):
print(rec["audio_path"], rec["num_channels"], rec["speaker_to_channel"])
# 4) Fetch a single clip's audio.
audio = hf_hub_download(REPO, rec["audio_path"], repo_type="dataset")
data, sr = sf.read(audio)
print(f"channels={data.ndim if data.ndim == 1 else data.shape[1]} sr={sr} frames={len(data)}")