diff --git a/.github/workflows/_run_notebook.yml b/.github/workflows/_run_notebook.yml index 607b8060..c8b466af 100644 --- a/.github/workflows/_run_notebook.yml +++ b/.github/workflows/_run_notebook.yml @@ -22,6 +22,19 @@ on: required: false type: string default: "{}" + # A notebook for something that has merged but not shipped pins the + # release that will carry it, and pip cannot install a version PyPI does + # not have yet. With this, the job says so and stops, instead of failing + # on an install nobody can fix from here. The alternative - installing + # espnet from git - would check something other than what a reader gets, + # which is the one thing tools/README.md says this must not do. + # The notebook's own workflow sets it and names the release it waits + # for; a run that is still skipping weeks later is a pin nobody + # finished. + allow_unreleased_pin: + required: false + type: boolean + default: false permissions: contents: read @@ -45,17 +58,38 @@ jobs: python-version: "3.12" - name: Install + id: install run: | set -euo pipefail python -m pip install --upgrade pip # the harness itself, before it can be asked anything python -m pip install nbclient nbformat ipykernel \ librosa scikit-learn matplotlib tqdm soundfile editdistance + # The release the notebook pins has to be on PyPI. It is not, for a + # notebook written between a feature merging and the release that + # carries it - see allow_unreleased_pin above. + asked=$(python tools/run_notebook.py \ + "${{ inputs.notebook }}" --print-install) + pinned=$(printf '%s\n' $asked \ + | grep -E '^espnet(\[[a-z,]+\])?==' || true) + version=${pinned##*==} + if [ -n "$version" ] && ! curl -sf -o /dev/null \ + "https://pypi.org/pypi/espnet/$version/json"; then + if [ "${{ inputs.allow_unreleased_pin }}" != "true" ]; then + echo "::error::this notebook pins espnet $version, which is" \ + "not on PyPI" + exit 1 + fi + echo "skip=true" >> "$GITHUB_OUTPUT" + echo "espnet $version is not on PyPI yet, so there is nothing to" \ + "install and nothing to run: this notebook is waiting for that" \ + "release." >> "$GITHUB_STEP_SUMMARY" + exit 0 + fi # then what this notebook's own pip lines ask for. The runner skips # those cells, so installing a different espnet from the one a # reader gets would make a pass here mean nothing. - python -m pip install $(python tools/run_notebook.py \ - "${{ inputs.notebook }}" --print-install) + python -m pip install $asked # and what it clones and installs - VERSA and ParallelWaveGAN are # on nobody's index. --no-build-isolation because ParallelWaveGAN's # setup.py imports pip, which an isolated build does not have. @@ -66,4 +100,5 @@ jobs: fi - name: Run ${{ inputs.notebook }} + if: steps.install.outputs.skip != 'true' run: python tools/run_notebook.py "${{ inputs.notebook }}" --timeout 4800 diff --git a/.github/workflows/s2t_align_demo.yml b/.github/workflows/s2t_align_demo.yml new file mode 100644 index 00000000..38238a0f --- /dev/null +++ b/.github/workflows/s2t_align_demo.yml @@ -0,0 +1,32 @@ +# Demos/s2t_align_demo.ipynb, every Sunday. The badge on this workflow is what +# the README shows beside that notebook, which is why it has a file of its own: +# GitHub's badge is per workflow, and cannot show one job of a matrix. + +name: s2t_align_demo + +on: + schedule: + # Sundays, 05:00 UTC + - cron: "0 5 * * 0" + workflow_dispatch: + pull_request: + paths: + - "Demos/s2t_align_demo.ipynb" + - "tools/**" + - ".github/workflows/s2t_align_demo.yml" + - ".github/workflows/_run_notebook.yml" + +permissions: + contents: read + +jobs: + run: + uses: ./.github/workflows/_run_notebook.yml + with: + notebook: Demos/s2t_align_demo.ipynb + # espnet2.bin.align merged after v.202610.post1 was tagged, so the + # release this notebook pins does not exist yet (espnet#6791 cuts it). + # Until then the job says so and skips. Delete this line when + # 202610.post2 is on PyPI - and if it is still here a month from now, + # the pin was never finished. + allow_unreleased_pin: true diff --git a/Demos/README.md b/Demos/README.md index 5f7d20df..c52ee9c8 100644 --- a/Demos/README.md +++ b/Demos/README.md @@ -39,6 +39,7 @@ the answer now. | [`asr_demo.ipynb`](asr_demo.ipynb) | [![asr_demo](https://github.com/espnet/notebook/actions/workflows/asr_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/asr_demo.yml) | Transcribe speech with OWSM-CTC, and let it work out the language | | [`asr_streaming_demo.ipynb`](asr_streaming_demo.ipynb) | [![asr_streaming_demo](https://github.com/espnet/notebook/actions/workflows/asr_streaming_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/asr_streaming_demo.yml) | Watch the words appear while the audio is still arriving | | [`st_demo.ipynb`](st_demo.ipynb) | [![st_demo](https://github.com/espnet/notebook/actions/workflows/st_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/st_demo.yml) | Translate English speech into German, French and Chinese — the same model | +| [`s2t_align_demo.ipynb`](s2t_align_demo.ipynb) | [![s2t_align_demo](https://github.com/espnet/notebook/actions/workflows/s2t_align_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/s2t_align_demo.yml) | Line text up with the audio it was said in, and score how well they agree | | [`tts_demo.ipynb`](tts_demo.ipynb) | [![tts_demo](https://github.com/espnet/notebook/actions/workflows/tts_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/tts_demo.yml) | Type a sentence, hear it spoken — one English voice, then 128 of them | | [`enh_demo.ipynb`](enh_demo.ipynb) | [![enh_demo](https://github.com/espnet/notebook/actions/workflows/enh_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/enh_demo.yml) | Pull speech out of noise, and measure how much it helped | | [`spk_demo.ipynb`](spk_demo.ipynb) | [![spk_demo](https://github.com/espnet/notebook/actions/workflows/spk_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/spk_demo.yml) | Turn a voice into a vector, and score two recordings against each other | diff --git a/Demos/s2t_align_demo.ipynb b/Demos/s2t_align_demo.ipynb new file mode 100644 index 00000000..fe0a4ee6 --- /dev/null +++ b/Demos/s2t_align_demo.ipynb @@ -0,0 +1,235 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "# Forced alignment — when each line was said\n", + "\n", + "[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/espnet/notebook/blob/master/Demos/s2t_align_demo.ipynb) [![s2t_align_demo](https://github.com/espnet/notebook/actions/workflows/s2t_align_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/s2t_align_demo.yml)\n", + "\n", + "You have a recording and the text of what was said. Alignment says *when*:\n", + "a start and an end for every line, a time for every token inside it, and a\n", + "score for how well the line and the audio agree. That is what makes\n", + "subtitles, what cuts a long recording into training utterances, and what\n", + "throws away the pairs where the text does not match the audio.\n", + "\n", + "Nothing is trained here: alignment reads the CTC head of a model that\n", + "already exists. CPU is enough. The checkpoint is 4 GB and cached after the\n", + "first run." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Install" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "%pip install -q \"espnet==202610.post2\" espnet_model_zoo librosa matplotlib" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## A recording, and what was said\n", + "\n", + "Four utterances, in the order they were spoken. Alignment needs the order;\n", + "it does not need the times, which is the whole point." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import librosa\n", + "from IPython.display import Audio, display\n", + "\n", + "!wget -q -O sample.wav https://github.com/espnet/espnet/raw/master/test_utils/ctc_align_test.wav\n", + "speech, rate = librosa.load(\"sample.wav\", sr=16000)\n", + "display(Audio(speech, rate=rate))\n", + "\n", + "utterances = [\n", + " \"The sale of the hotels\",\n", + " \"is part of Holiday's strategy\",\n", + " \"to sell off assets\",\n", + " \"and concentrate on property management\",\n", + "]\n", + "print(f\"{len(speech) / rate:.1f} s of audio, {len(utterances)} utterances\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Aligning\n", + "\n", + "`ForcedAligner` takes any published checkpoint with a CTC head — an OWSM\n", + "one here, but an ASR model works the same way — and returns one segment an\n", + "utterance: when it starts, when it ends, and how sure the model is." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from espnet2.bin.align import ForcedAligner\n", + "\n", + "aligner = ForcedAligner.from_pretrained(\"espnet/owsm_ctc_v4_1B\", device=\"cpu\")\n", + "segments = aligner(\"sample.wav\", utterances)\n", + "\n", + "for s in segments:\n", + " print(f\"{s.start:5.2f} {s.end:5.2f} {s.score:.3f} {s.text}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Inside an utterance\n", + "\n", + "Each segment carries the tokens it was made of, with a time and a\n", + "probability each. Word-level timestamps come from here." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "for token in segments[0].tokens:\n", + " print(f\"{token.start:5.2f} {token.end:5.2f} {token.score:.3f} {token.text}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Seeing it" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import numpy as np\n", + "\n", + "fig, ax = plt.subplots(figsize=(11, 3))\n", + "ax.plot(np.arange(len(speech)) / rate, speech, linewidth=0.4, color=\"#888\")\n", + "for i, s in enumerate(segments):\n", + " ax.axvspan(s.start, s.end, color=f\"C{i}\", alpha=0.25)\n", + " ax.text(\n", + " (s.start + s.end) / 2,\n", + " 0.85 * speech.max(),\n", + " s.text.split()[0],\n", + " ha=\"center\",\n", + " fontsize=8,\n", + " )\n", + "ax.set_xlabel(\"seconds\")\n", + "ax.set_yticks([])\n", + "plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## The score is the point\n", + "\n", + "`score` is the mean probability of an utterance's tokens: 1.0 is a perfect\n", + "match. Text that does not belong to the audio still gets aligned — the\n", + "algorithm has to put it somewhere — and the score is how you find out.\n", + "\n", + "This is alignment-score filtering, which is how OWSM v4 cleaned its training\n", + "data: align, then drop the pairs that score badly\n", + "([recipe](https://github.com/espnet/espnet/tree/master/egs2/owsm_v4/s2t1))." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "wrong = list(utterances)\n", + "wrong[1] = \"the weather today is cold\"\n", + "\n", + "for s in aligner(\"sample.wav\", wrong):\n", + " print(f\"{s.score:.3f} {s.text}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Write the text the way the model writes it\n", + "\n", + "The score is a probability under *this* model, so the text has to be spelled\n", + "the way the model spells it. The same four sentences in capitals — which is\n", + "how this recording's reference transcript is written — score zero, because\n", + "OWSM's vocabulary has no capitalised words and each one breaks into letters.\n", + "The times are still roughly right; the score is not." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "shouted = [u.upper() for u in utterances]\n", + "loud = aligner(\"sample.wav\", shouted)\n", + "\n", + "for s in loud:\n", + " print(f\"{s.score:.3f} {s.text}\")\n", + "\n", + "# the first line, as the model reads it\n", + "print([t.text for t in loud[0].tokens[:8]])" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Where next\n", + "\n", + "- **From the terminal**: `espnet align sample.wav --text \"The sale of the hotels\"`\n", + "- **From an assistant**: the MCP server offers the same thing as an `align` tool\n", + "- **Transcription** with the same checkpoint: [`asr_demo.ipynb`](asr_demo.ipynb)\n", + "- **The other algorithm**: `espnet2.bin.ctc_segment` is the\n", + " [`ctc_segmentation`](https://arxiv.org/abs/2007.09127) package's, which\n", + " partitions the timeline rather than marking where each token is, and scores\n", + " in log space. `espnet2/bin/asr_align.py` and `s2t_align.py` are its scripts." + ] + } + ], + "metadata": { + "colab": { + "provenance": [] + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + }, + "language_info": { + "name": "python" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/README.md b/README.md index 33017eee..69d7cc3a 100644 --- a/README.md +++ b/README.md @@ -17,6 +17,7 @@ One per task, flat in [`Demos/`](Demos), each short enough to read in a sitting. | [`asr_demo.ipynb`](Demos/asr_demo.ipynb) | [![asr_demo](https://github.com/espnet/notebook/actions/workflows/asr_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/asr_demo.yml) | Transcribe speech with OWSM-CTC, and let it work out the language | | [`asr_streaming_demo.ipynb`](Demos/asr_streaming_demo.ipynb) | [![asr_streaming_demo](https://github.com/espnet/notebook/actions/workflows/asr_streaming_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/asr_streaming_demo.yml) | Watch the words appear while the audio is still arriving | | [`st_demo.ipynb`](Demos/st_demo.ipynb) | [![st_demo](https://github.com/espnet/notebook/actions/workflows/st_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/st_demo.yml) | Translate English speech into German, French and Chinese — the same model | +| [`s2t_align_demo.ipynb`](Demos/s2t_align_demo.ipynb) | [![s2t_align_demo](https://github.com/espnet/notebook/actions/workflows/s2t_align_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/s2t_align_demo.yml) | Line text up with the audio it was said in, and score how well they agree | | [`tts_demo.ipynb`](Demos/tts_demo.ipynb) | [![tts_demo](https://github.com/espnet/notebook/actions/workflows/tts_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/tts_demo.yml) | Type a sentence, hear it spoken — one English voice, then 128 of them | | [`enh_demo.ipynb`](Demos/enh_demo.ipynb) | [![enh_demo](https://github.com/espnet/notebook/actions/workflows/enh_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/enh_demo.yml) | Pull speech out of noise, and measure how much it helped | | [`spk_demo.ipynb`](Demos/spk_demo.ipynb) | [![spk_demo](https://github.com/espnet/notebook/actions/workflows/spk_demo.yml/badge.svg)](https://github.com/espnet/notebook/actions/workflows/spk_demo.yml) | Turn a voice into a vector, and score two recordings against each other | diff --git a/tools/README.md b/tools/README.md index 2ad9fe13..873d96f6 100644 --- a/tools/README.md +++ b/tools/README.md @@ -47,6 +47,29 @@ the install cell an interface, and this is what it supports. | `!apt-get install …`, `!cd repo && pip install .` | removed before the notebook is executed, because the runner is not Colab, but not treated as a Python dependency | | anything in a markdown cell | it is prose | +### Pinned to a release that is not out yet + +A demo for something that has merged but has not shipped pins the release that +will carry it, and pip cannot install a version PyPI does not have. The +notebook's workflow says so: + +```yaml + with: + notebook: Demos/s2t_align_demo.ipynb + # espnet2.bin.align merged after v.202610.post1 was tagged + allow_unreleased_pin: true +``` + +and the run stops after installing the harness, with a line in the job summary +saying which release it is waiting for, rather than failing on an install +nobody can fix from here. Without the flag, a pin PyPI does not have is an +error naming the version - a typo should not pass quietly. + +What this does **not** do is install espnet from git, for the reason in the +table above: a pass here would then say nothing about what a reader gets. The +line is temporary by construction. A run that is still skipping weeks later is +a pin nobody finished, and the badge is a skip rather than a green tick. + ### Cloned and installed A tool that is on nobody's index is fetched in two shell lines: