Spaces:

Tamazight-NLP
/

ASR-N-gram-Language-modeling

Running

App Files Files Community

ayymen commited on Oct 24

Commit

143d264

•

1 Parent(s): afb069e

Initial commit

Browse files

Files changed (10) hide show

135.wav +0 -0
Dockerfile +54 -0
README.md +1 -1
app.py +185 -0
common_voice_zgh_37837257.mp3 +0 -0
install_beamsearch_decoders.sh +70 -0
kenlm.bin +3 -0
packages.txt +2 -0
pre-requirements.txt +1 -0
requirements.txt +7 -0

135.wav ADDED Viewed

Binary file (171 kB). View file

Dockerfile ADDED Viewed

	@@ -0,0 +1,54 @@

+#!/usr/bin/env bash
+# Copyright (c) 2023, NVIDIA CORPORATION.  All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# Use this script to install KenLM, OpenSeq2Seq decoder, Flashlight decoder, OpenGRM Ngram tool to contaner
+# How to use? Build it from NeMo root folder:
+# 1. git clone https://github.com/NVIDIA/NeMo.git && cd NeMo
+# 2. DOCKER_BUILDKIT=1 docker build -t nemo:23.03.1 -f ./scripts/installers/Dockerfile.ngramtools .
+FROM nvcr.io/nvidia/nemo:24.01.speech
+WORKDIR /workspace/nemo
+COPY ./install_beamsearch_decoders.sh /workspace/nemo/scripts/asr_language_modeling/ngram_lm/install_beamsearch_decoders.sh
+RUN /bin/bash scripts/asr_language_modeling/ngram_lm/install_beamsearch_decoders.sh
+RUN --mount=target=/tmp/packages.txt,source=packages.txt 	apt-get update && 	xargs -r -a /tmp/packages.txt apt-get install -y && 	rm -rf /var/lib/apt/lists/*
+RUN --mount=target=/tmp/pre-requirements.txt,source=pre-requirements.txt 	pip install --no-cache-dir -r /tmp/pre-requirements.txt
+RUN --mount=target=/tmp/requirements.txt,source=requirements.txt     pip install --no-cache-dir -r /tmp/requirements.txt
+WORKDIR /code
+# Set up a new user named "user" with user ID 1000
+RUN useradd -m -u 1000 user
+# Switch to the "user" user
+USER user
+# Set home to the user's home directory
+ENV HOME=/home/user \
+    PATH=/home/user/.local/bin:$PATH
+# Set the working directory to the user's home directory
+WORKDIR $HOME/app
+# Copy the current directory contents into the container at $HOME/app setting the owner to the user
+COPY --chown=user . $HOME/app
+CMD ["python", "app.py"]

README.md CHANGED Viewed

@@ -1,5 +1,5 @@
 ---
-title: ASR N Gram Language Modeling
 emoji: 👀
 colorFrom: purple
 colorTo: gray

 ---
+title: ASR N-gram Language Modeling
 emoji: 👀
 colorFrom: purple
 colorTo: gray

app.py ADDED Viewed

	@@ -0,0 +1,185 @@

+from nemo.collections.asr.models import EncDecCTCModelBPE
+from omegaconf import open_dict
+#import yt_dlp as youtube_dl
+import os
+import tempfile
+import torch
+import gradio as gr
+from pydub import AudioSegment
+import time
+device = "cuda" if torch.cuda.is_available() else "cpu"
+MODEL_NAME="ayymen/stt_zgh_fastconformer_ctc_small"
+YT_LENGTH_LIMIT_S=3600
+model = EncDecCTCModelBPE.from_pretrained(model_name=MODEL_NAME).to(device)
+with open_dict(model.cfg):
+    model.cfg.decoding.strategy = "beam"
+    model.cfg.decoding.beam.beam_size = 256 # Desired Beam Size
+    model.cfg.decoding.beam.beam_alpha = 1.5 # Desired Beam Alpha
+    model.cfg.decoding.beam.beam_beta = 1.5 # Desired Beam Beta
+    model.cfg.decoding.beam.kenlm_path = "kenlm.bin" # Path to KenLM binary file
+model.change_decoding_strategy(model.cfg.decoding)
+model.eval()
+def get_transcripts(audio_path):
+    audio = AudioSegment.from_file(audio_path)
+    # check if audio is mono 16kHz
+    if audio.channels != 1 or audio.frame_rate != 16000:
+        audio = audio.set_channels(1).set_frame_rate(16000) # convert to mono 16kHz
+        with tempfile.TemporaryDirectory() as tmpdirname:
+            audio_path = os.path.join(tmpdirname, "audio.wav")
+            audio.export(audio_path, format="wav")
+            text = model.transcribe([audio_path])[0]
+    else:
+        text = model.transcribe([audio_path])[0]
+    return text
+'''
+article = (
+    "<p style='text-align: center'>"
+    "<a href='https://huggingface.co/nvidia/parakeet-rnnt-1.1b' target='_blank'>🎙️ Learn more about Parakeet model</a> | "
+    "<a href='https://arxiv.org/abs/2305.05084' target='_blank'>📚 FastConformer paper</a> | "
+    "<a href='https://github.com/NVIDIA/NeMo' target='_blank'>🧑‍💻 Repository</a>"
+    "</p>"
+)
+'''
+EXAMPLES = [
+    ["135.wav"],
+    ["common_voice_zgh_37837257.mp3"]
+]
+"""
+YT_EXAMPLES = [
+    ["https://www.youtube.com/shorts/CSgTSE50MHY"],
+    ["https://www.youtube.com/shorts/OxQtqOyAFLE"]
+]
+"""
+def _return_yt_html_embed(yt_url):
+    video_id = yt_url.split("?v=")[-1]
+    if "youtube.com/shorts/" in video_id:
+        video_id = video_id.split("/")[-1]
+    HTML_str = (
+        f'<center> <iframe width="500" height="320" src="https://www.youtube.com/embed/{video_id}"> </iframe>'
+        " </center>"
+    )
+    return HTML_str
+def download_yt_audio(yt_url, filename):
+    info_loader = youtube_dl.YoutubeDL()
+    try:
+        info = info_loader.extract_info(yt_url, download=False)
+    except youtube_dl.utils.DownloadError as err:
+        raise gr.Error(str(err))
+    file_length = info["duration_string"]
+    file_h_m_s = file_length.split(":")
+    file_h_m_s = [int(sub_length) for sub_length in file_h_m_s]
+    if len(file_h_m_s) == 1:
+        file_h_m_s.insert(0, 0)
+    if len(file_h_m_s) == 2:
+        file_h_m_s.insert(0, 0)
+    file_length_s = file_h_m_s[0] * 3600 + file_h_m_s[1] * 60 + file_h_m_s[2]
+    if file_length_s > YT_LENGTH_LIMIT_S:
+        yt_length_limit_hms = time.strftime("%HH:%MM:%SS", time.gmtime(YT_LENGTH_LIMIT_S))
+        file_length_hms = time.strftime("%HH:%MM:%SS", time.gmtime(file_length_s))
+        raise gr.Error(f"Maximum YouTube length is {yt_length_limit_hms}, got {file_length_hms} YouTube video.")
+    ydl_opts = {"outtmpl": filename, "format": "worstvideo[ext=mp4]+bestaudio[ext=m4a]/best[ext=mp4]/best"}
+    with youtube_dl.YoutubeDL(ydl_opts) as ydl:
+        try:
+            ydl.download([yt_url])
+        except youtube_dl.utils.ExtractorError as err:
+            raise gr.Error(str(err))
+def yt_transcribe(yt_url, max_filesize=75.0):
+    html_embed_str = _return_yt_html_embed(yt_url)
+    with tempfile.TemporaryDirectory() as tmpdirname:
+        filepath = os.path.join(tmpdirname, "video.mp4")
+        download_yt_audio(yt_url, filepath)
+        audio = AudioSegment.from_file(filepath)
+        audio = audio.set_channels(1).set_frame_rate(16000) # convert to mono 16kHz
+        wav_filepath = os.path.join(tmpdirname, "audio.wav")
+        audio.export(wav_filepath, format="wav")
+        text = get_transcripts(wav_filepath)
+    return html_embed_str, text
+demo = gr.Blocks()
+mf_transcribe = gr.Interface(
+    fn=get_transcripts,
+    inputs=[
+        gr.Audio(sources="microphone", type="filepath")
+    ],
+    outputs="text",
+    title="Transcribe Audio",
+    description=(
+        "Transcribe microphone or audio inputs with the click of a button! Demo uses the"
+        f" checkpoint [{MODEL_NAME}](https://huggingface.co/{MODEL_NAME}) and [NVIDIA NeMo](https://github.com/NVIDIA/NeMo) to transcribe audio files"
+        " of arbitrary length."
+    ),
+    allow_flagging="never",
+)
+file_transcribe = gr.Interface(
+    fn=get_transcripts,
+    inputs=[
+        gr.Audio(sources="upload", type="filepath", label="Audio file"),
+    ],
+    outputs="text",
+    examples=EXAMPLES,
+    title="Transcribe Audio",
+    description=(
+        "Transcribe microphone or audio inputs with the click of a button! Demo uses the"
+        f" checkpoint [{MODEL_NAME}](https://huggingface.co/{MODEL_NAME}) and [NVIDIA NeMo](https://github.com/NVIDIA/NeMo) to transcribe audio files"
+        " of arbitrary length."
+    ),
+    allow_flagging="never",
+)
+"""
+youtube_transcribe = gr.Interface(
+    fn=yt_transcribe,
+    inputs=[
+        gr.Textbox(lines=1, placeholder="Paste the URL to a YouTube video here", label="YouTube URL"),
+    ],
+    outputs=["html", "text"],
+    examples=YT_EXAMPLES,
+    title="Transcribe Audio",
+    description=(
+        "Transcribe microphone or audio inputs with the click of a button! Demo uses the"
+        f" checkpoint [{MODEL_NAME}](https://huggingface.co/{MODEL_NAME}) and [NVIDIA NeMo](https://github.com/NVIDIA/NeMo) to transcribe audio files"
+        " of arbitrary length."
+    ),
+    allow_flagging="never",
+)
+"""
+with demo:
+    gr.TabbedInterface(
+        [
+            mf_transcribe,
+            file_transcribe,
+            #youtube_transcribe
+        ],
+        [
+            "Microphone",
+            "Audio file",
+            #"Youtube Video"
+        ]
+    )
+demo.launch()

common_voice_zgh_37837257.mp3 ADDED Viewed

Binary file (28.1 kB). View file

install_beamsearch_decoders.sh ADDED Viewed

	@@ -0,0 +1,70 @@

+#!/usr/bin/env bash
+# Copyright (c) 2022, NVIDIA CORPORATION.  All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# Use this script to install KenLM, OpenSeq2Seq decoder, Flashlight decoder
+shopt -s expand_aliases
+NEMO_PATH=/workspace/nemo  # Path to NeMo folder: /workspace/nemo if you use NeMo/Dockerfile
+if [ "$#" -eq 1 ]; then
+  NEMO_PATH=$1
+fi
+KENLM_MAX_ORDER=10 # Maximum order of KenLM model, also specified in the setup_os2s_decoders.py
+if [ -d "$NEMO_PATH" ]; then
+  echo "The folder '$NEMO_PATH' exists."
+else
+  echo "Error: The folder '$NEMO_PATH' does not exist. Specify it as a first command line positional argument!"
+  exit 1
+fi
+cd $NEMO_PATH
+if [ $(id -u) -eq 0 ]; then
+  alias aptupdate='apt-get update'
+  alias b2install='./b2'
+else
+  alias aptupdate='sudo apt-get update'
+  alias b2install='sudo ./b2'
+fi
+aptupdate && apt-get upgrade -y
+# apt-get install -y swig liblzma-dev && rm -rf /var/lib/apt/lists/* # liblzma needed for flashlight decoder
+# install Boost package for KenLM
+wget https://boostorg.jfrog.io/artifactory/main/release/1.80.0/source/boost_1_80_0.tar.bz2 --no-check-certificate && tar --bzip2 -xf $NEMO_PATH/boost_1_80_0.tar.bz2 && cd boost_1_80_0 && ./bootstrap.sh && b2install --layout=tagged link=static,shared threading=multi,single install -j4 && cd .. || echo FAILURE
+export BOOST_ROOT=$NEMO_PATH/boost_1_80_0
+git clone https://github.com/NVIDIA/OpenSeq2Seq
+cd OpenSeq2Seq
+git checkout ctc-decoders
+cd ..
+mv OpenSeq2Seq/decoders $NEMO_PATH/
+rm -rf OpenSeq2Seq
+cd $NEMO_PATH/decoders
+cp $NEMO_PATH/scripts/installers/setup_os2s_decoders.py ./setup.py
+./setup.sh
+# install KenLM
+cd $NEMO_PATH/decoders/kenlm/build && cmake -DKENLM_MAX_ORDER=$KENLM_MAX_ORDER .. && make -j2
+cd $NEMO_PATH/decoders/kenlm
+python setup.py install --max_order=$KENLM_MAX_ORDER
+export KENLM_LIB=$NEMO_PATH/decoders/kenlm/build/bin
+export KENLM_ROOT=$NEMO_PATH/decoders/kenlm
+cd ..
+# install Flashlight
+# git clone https://github.com/flashlight/text && cd text
+# python setup.py bdist_wheel
+# pip install dist/*.whl
+# cd ..

kenlm.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:5e3106cf41031192efb8bd1c615f2f40fc12b9d9ae7132b5b6e16b50aa8e4b83
+size 69178189

packages.txt ADDED Viewed

	@@ -0,0 +1,2 @@


1	+ ffmpeg
2	+ libsndfile1

pre-requirements.txt ADDED Viewed

	@@ -0,0 +1 @@


1	+ Cython

requirements.txt ADDED Viewed

	@@ -0,0 +1,7 @@

+Cython
+huggingface-hub==0.23.2
+# nemo-toolkit[asr]==2.0.0rc1
+# numpy<2.0.0
+# ipython
+# yt_dlp
+gradio