"""Export FLORES-200 zh↔en into line-aligned src/ref files for eval_mt. Corpus is not committed. Typical: uv run python scripts/export_flores.py --out data/eval """ from __future__ import annotations import argparse from pathlib import Path def _write(path: Path, lines: list[str]) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text("\n".join(lines) + "\n", encoding="utf-8") def main() -> None: p = argparse.ArgumentParser(description=__doc__) p.add_argument("--out", default="data/eval") p.add_argument("--split", default="devtest", choices=["dev", "devtest"]) args = p.parse_args() from datasets import load_dataset ds = None err: Exception | None = None for config in ("eng_Latn-zho_Hans", "default"): try: ds = load_dataset("facebook/flores", config, split=args.split) break except Exception as exc: # noqa: BLE001 — try the next config name err = exc if ds is None: raise SystemExit(f"could not load facebook/flores ({err})") cols = set(ds.column_names) en_key = next( (c for c in ("sentence_eng_Latn", "eng_Latn", "sentence_en") if c in cols), None, ) zh_key = next( (c for c in ("sentence_zho_Hans", "zho_Hans", "sentence_zh") if c in cols), None, ) if en_key is None or zh_key is None: raise SystemExit(f"FLORES columns not found: {sorted(cols)}") en = [row[en_key].strip() for row in ds] zh = [row[zh_key].strip() for row in ds] out = Path(args.out) _write(out / "flores.zh2en.src.txt", zh) _write(out / "flores.zh2en.ref.txt", en) _write(out / "flores.en2zh.src.txt", en) _write(out / "flores.en2zh.ref.txt", zh) print(f"wrote {len(zh)} pairs under {out}/flores.*.txt") if __name__ == "__main__": main()