k2-fsa
diff --git a/‎.github/workflows/run-pretrained-transducer-stateless.yml
+108 b/‎.github/workflows/run-pretrained-transducer-stateless.yml
+108
diff --git a/‎.github/workflows/run-pretrained.yml
+1-1 b/‎.github/workflows/run-pretrained.yml
+1-1
diff --git a/‎.github/workflows/test.yml
+13-1 b/‎.github/workflows/test.yml
+13-1
diff --git a/‎README.md
+21-4 b/‎README.md
+21-4
diff --git a/‎egs/librispeech/ASR/README.md
+17 b/‎egs/librispeech/ASR/README.md
+17
diff --git a/‎egs/librispeech/ASR/RESULTS.md
+62-3 b/‎egs/librispeech/ASR/RESULTS.md
+62-3
diff --git a/‎egs/librispeech/ASR/transducer/model.py
+5-5 b/‎egs/librispeech/ASR/transducer/model.py
+5-5
diff --git a/‎egs/librispeech/ASR/transducer_lstm/model.py
+5-5 b/‎egs/librispeech/ASR/transducer_lstm/model.py
+5-5
@@ -0,0 +1,108 @@
+# Copyright      2021  Fangjun Kuang (csukuangfj@gmail.com)
+
+# See ../../LICENSE for clarification regarding multiple authors
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+name: run-pre-trained-tranducer-stateless
+
+on:
+  push:
+    branches:
+      - master
+  pull_request:
+    types: [labeled]
+
+jobs:
+  run_pre_trained_transducer_stateless:
+    if: github.event.label.name == 'ready' || github.event_name == 'push'
+    runs-on: ${{ matrix.os }}
+    strategy:
+      matrix:
+        os: [ubuntu-18.04]
+        python-version: [3.7, 3.8, 3.9]
+        torch: ["1.10.0"]
+        torchaudio: ["0.10.0"]
+        k2-version: ["1.9.dev20211101"]
+
+      fail-fast: false
+
+    steps:
+      - uses: actions/checkout@v2
+        with:
+          fetch-depth: 0
+
+      - name: Setup Python ${{ matrix.python-version }}
+        uses: actions/setup-python@v1
+        with:
+          python-version: ${{ matrix.python-version }}
+
+      - name: Install Python dependencies
+        run: |
+          python3 -m pip install --upgrade pip pytest
+          # numpy 1.20.x does not support python 3.6
+          pip install numpy==1.19
+          pip install torch==${{ matrix.torch }}+cpu torchaudio==${{ matrix.torchaudio }}+cpu -f https://download.pytorch.org/whl/cpu/torch_stable.html
+          pip install k2==${{ matrix.k2-version }}+cpu.torch${{ matrix.torch }} -f https://k2-fsa.org/nightly/
+
+          python3 -m pip install git+https://github.com/lhotse-speech/lhotse
+          python3 -m pip install kaldifeat
+          # We are in ./icefall and there is a file: requirements.txt in it
+          pip install -r requirements.txt
+
+      - name: Install graphviz
+        shell: bash
+        run: |
+          python3 -m pip install -qq graphviz
+          sudo apt-get -qq install graphviz
+
+      - name: Download pre-trained model
+        shell: bash
+        run: |
+          sudo apt-get -qq install git-lfs tree sox
+          cd egs/librispeech/ASR
+          mkdir tmp
+          cd tmp
+          git lfs install
+          git clone https://huggingface.co/csukuangfj/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22
+          cd ..
+          tree tmp
+          soxi tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/test_wavs/*.wav
+          ls -lh tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/test_wavs/*.wav
+
+      - name: Run greedy search decoding
+        shell: bash
+        run: |
+          export PYTHONPATH=$PWD:PYTHONPATH
+          cd egs/librispeech/ASR
+          ./transducer_stateless/pretrained.py \
+            --method greedy_search \
+            --checkpoint ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/exp/pretrained.pt \
+            --bpe-model ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/data/lang_bpe_500/bpe.model \
+            ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/test_wavs/1089-134686-0001.wav \
+            ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/test_wavs/1221-135766-0001.wav \
+            ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/test_wavs/1221-135766-0002.wav
+
+      - name: Run beam search decoding
+        shell: bash
+        run: |
+          export PYTHONPATH=$PWD:$PYTHONPATH
+          cd egs/librispeech/ASR
+          ./transducer_stateless/pretrained.py \
+            --method beam_search \
+            --beam-size 4 \
+            --checkpoint ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/exp/pretrained.pt \
+            --bpe-model ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/data/lang_bpe_500/bpe.model \
+            ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/test_wavs/1089-134686-0001.wav \
+            ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/test_wavs/1221-135766-0001.wav \
+            ./tmp/icefall-asr-librispeech-transducer-stateless-bpe-500-2021-12-22/test_wavs/1221-135766-0002.wav
@@ -30,7 +30,7 @@ jobs:
     strategy:
       matrix:
         os: [ubuntu-18.04]
-        python-version: [3.6, 3.7, 3.8, 3.9]
+        python-version: [3.7, 3.8, 3.9]
         torch: ["1.10.0"]
         torchaudio: ["0.10.0"]
         k2-version: ["1.9.dev20211101"]
 
@@ -32,7 +32,7 @@ jobs:
         # os: [ubuntu-18.04, macos-10.15]
         # disable macOS test for now.
         os: [ubuntu-18.04]
-        python-version: [3.6, 3.7, 3.8, 3.9]
+        python-version: [3.7, 3.8]
         torch: ["1.8.0", "1.10.0"]
         torchaudio: ["0.8.0", "0.10.0"]
         k2-version: ["1.9.dev20211101"]
@@ -106,6 +106,12 @@ jobs:
           if [[ ${{ matrix.torchaudio }} == "0.10.0" ]]; then
             cd ../transducer
             pytest -v -s
+
+            cd ../transducer_stateless
+            pytest -v -s
+
+            cd ../transducer_lstm
+            pytest -v -s
           fi
 
       - name: Run tests
@@ -125,4 +131,10 @@ jobs:
           if [[ ${{ matrix.torchaudio }} == "0.10.0" ]]; then
             cd ../transducer
             pytest -v -s
+
+            cd ../transducer_stateless
+            pytest -v -s
+
+            cd ../transducer_lstm
+            pytest -v -s
           fi
@@ -34,11 +34,12 @@ We do provide a Colab notebook for this recipe.
 
 ### LibriSpeech
 
-We provide 3 models for this recipe:
+We provide 4 models for this recipe:
 
 - [conformer CTC model][LibriSpeech_conformer_ctc]
 - [TDNN LSTM CTC model][LibriSpeech_tdnn_lstm_ctc]
-- [RNN-T Conformer model][LibriSpeech_transducer]
+- [Transducer: Conformer encoder + LSTM decoder][LibriSpeech_transducer]
+- [Transducer: Conformer encoder + Embedding decoder][LibriSpeech_transducer_stateless]
 
 #### Conformer CTC Model
 
@@ -62,9 +63,9 @@ The WER for this model is:
 We provide a Colab notebook to run a pre-trained TDNN LSTM CTC model:  [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/drive/1kNmDXNMwREi0rZGAOIAOJo93REBuOTcd?usp=sharing)
 
 
-#### RNN-T Conformer model
+#### Transducer: Conformer encoder + LSTM decoder
 
-Using Conformer as encoder.
+Using Conformer as encoder and LSTM as decoder.
 
 The best WER with greedy search is:
 
@@ -74,6 +75,21 @@ The best WER with greedy search is:
 
 We provide a Colab notebook to run a pre-trained RNN-T conformer model: [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/drive/1_u6yK9jDkPwG_NLrZMN2XK7Aeq4suMO2?usp=sharing)
 
+#### Transducer: Conformer encoder + Embedding decoder
+
+Using Conformer as encoder. The decoder consists of 1 embedding layer
+and 1 convolutional layer.
+
+The best WER using beam search with beam size 4 is:
+
+|     | test-clean | test-other |
+|-----|------------|------------|
+| WER | 2.92       | 7.37       |
+
+Note: No auxiliary losses are used in the training and no LMs are used
+in the decoding.
+
+We provide a Colab notebook to run a pre-trained transducer conformer + stateless decoder model: [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/drive/1Lm37sNajIpkV4HTzMDF7sn9l0JpfmekN?usp=sharing)
 
 ### Aishell
 
@@ -143,6 +159,7 @@ Please see: [![Open In Colab](https://colab.research.google.com/assets/colab-bad
 [LibriSpeech_tdnn_lstm_ctc]: egs/librispeech/ASR/tdnn_lstm_ctc
 [LibriSpeech_conformer_ctc]: egs/librispeech/ASR/conformer_ctc
 [LibriSpeech_transducer]: egs/librispeech/ASR/transducer
+[LibriSpeech_transducer_stateless]: egs/librispeech/ASR/transducer_stateless
 [Aishell_tdnn_lstm_ctc]: egs/aishell/ASR/tdnn_lstm_ctc
 [Aishell_conformer_ctc]: egs/aishell/ASR/conformer_ctc
 [TIMIT_tdnn_lstm_ctc]: egs/timit/ASR/tdnn_lstm_ctc
 
@@ -1,3 +1,20 @@
 
+# Introduction
+
 Please refer to <https://icefall.readthedocs.io/en/latest/recipes/librispeech.html>
 for how to run models in this recipe.
+
+# Transducers
+
+There are various folders containing the name `transducer` in this folder.
+The following table lists the differences among them.
+
+|                        | Encoder   | Decoder            |
+|------------------------|-----------|--------------------|
+| `transducer`           | Conformer | LSTM               |
+| `transducer_stateless` | Conformer | Embedding + Conv1d |
+| `transducer_lstm     ` | LSTM      | LSTM               |
+
+The decoder in `transducer_stateless` is modified from the paper
+[Rnn-Transducer with Stateless Prediction Network](https://ieeexplore.ieee.org/document/9054419/).
+We place an additional Conv1d layer right after the input embedding layer.
@@ -1,18 +1,77 @@
 ## Results
 
-### LibriSpeech BPE training results (RNN-T)
+### LibriSpeech BPE training results (Transducer)
+
+#### 2021-12-22
+Conformer encoder + non-current decoder. The decoder
+contains only an embedding layer and a Conv1d (with kernel size 2).
+
+The WERs are
+
+|                           | test-clean | test-other | comment                                  |
+|---------------------------|------------|------------|------------------------------------------|
+| greedy search             | 2.99       | 7.52       | --epoch 20, --avg 10, --max-duration 100 |
+| beam search (beam size 2) | 2.95       | 7.43       |                                          |
+| beam search (beam size 3) | 2.94       | 7.37       |                                          |
+| beam search (beam size 4) | 2.92       | 7.37       |                                          |
+| beam search (beam size 5) | 2.93       | 7.38       |                                          |
+| beam search (beam size 8) | 2.92       | 7.38       |                                          |
+
+The training command for reproducing is given below:
+
+```
+export CUDA_VISIBLE_DEVICES="0,1,2,3"
+
+./transducer_stateless/train.py \
+  --world-size 4 \
+  --num-epochs 30 \
+  --start-epoch 0 \
+  --exp-dir transducer_stateless/exp-full \
+  --full-libri 1 \
+  --max-duration 250 \
+  --lr-factor 3
+```
+
+The tensorboard training log can be found at
+<https://tensorboard.dev/experiment/PsJ3LgkEQfOmzedAlYfVeg/#scalars&_smoothingWeight=0>
+
+The decoding command is:
+```
+epoch=20
+avg=10
+
+## greedy search
+./transducer_stateless/decode.py \
+  --epoch $epoch \
+  --avg $avg \
+  --exp-dir transducer_stateless/exp-full \
+  --bpe-model ./data/lang_bpe_500/bpe.model \
+  --max-duration 100
+
+## beam search
+./transducer_stateless/decode.py \
+  --epoch $epoch \
+  --avg $avg \
+  --exp-dir transducer_stateless/exp-full \
+  --bpe-model ./data/lang_bpe_500/bpe.model \
+  --max-duration 100 \
+  --decoding-method beam_search \
+  --beam-size 4
+```
+
 
 #### 2021-12-17
+Using commit `cb04c8a7509425ab45fae888b0ca71bbbd23f0de`.
 
-RNN-T + Conformer encoder
+Conformer encoder + LSTM decoder.
 
 The best WER is
 
 |     | test-clean | test-other |
 |-----|------------|------------|
 | WER | 3.16       | 7.71       |
 
-using `--epoch 26 --avg 12` during decoding with greedy search.
+using `--epoch 26 --avg 12` with **greedy search**.
 
 The training command to reproduce the above WER is:
 
 
@@ -27,11 +27,6 @@
 
 from icefall.utils import add_sos
 
-assert hasattr(torchaudio.functional, "rnnt_loss"), (
-    f"Current torchaudio version: {torchaudio.__version__}\n"
-    "Please install a version >= 0.10.0"
-)
-
 
 class Transducer(nn.Module):
     """It implements https://arxiv.org/pdf/1211.3711.pdf
@@ -115,6 +110,11 @@ def forward(
         # Note: y does not start with SOS
         y_padded = y.pad(mode="constant", padding_value=0)
 
+        assert hasattr(torchaudio.functional, "rnnt_loss"), (
+            f"Current torchaudio version: {torchaudio.__version__}\n"
+            "Please install a version >= 0.10.0"
+        )
+
         loss = torchaudio.functional.rnnt_loss(
             logits=logits,
             targets=y_padded,
 
@@ -27,11 +27,6 @@
 
 from icefall.utils import add_sos
 
-assert hasattr(torchaudio.functional, "rnnt_loss"), (
-    f"Current torchaudio version: {torchaudio.__version__}\n"
-    "Please install a version >= 0.10.0"
-)
-
 
 class Transducer(nn.Module):
     """It implements https://arxiv.org/pdf/1211.3711.pdf
@@ -115,6 +110,11 @@ def forward(
         # Note: y does not start with SOS
         y_padded = y.pad(mode="constant", padding_value=0)
 
+        assert hasattr(torchaudio.functional, "rnnt_loss"), (
+            f"Current torchaudio version: {torchaudio.__version__}\n"
+            "Please install a version >= 0.10.0"
+        )
+
         loss = torchaudio.functional.rnnt_loss(
             logits=logits,
             targets=y_padded,