From 819ab35b284abb8e4afbca3a0bd5b03c78b6f934 Mon Sep 17 00:00:00 2001 From: makaveli10 Date: Fri, 24 May 2024 04:58:49 -0400 Subject: [PATCH] fix: limit CPU usage for VAD onnxruntime inference session by setting OMP_NUM_THREADS Signed-off-by: makaveli10 --- README.md | 8 +++++++- run_server.py | 10 +++++++++- 2 files changed, 16 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index d4c8460..d52ef9a 100644 --- a/README.md +++ b/README.md @@ -53,7 +53,13 @@ python3 run_server.py -p 9090 \ -trt /home/TensorRT-LLM/examples/whisper/whisper_small \ -m ``` - +#### Controlling OpenMP Threads +To control the number of threads used by OpenMP, you can set the `OMP_NUM_THREADS` environment variable. This is useful for managing CPU resources and ensuring consistent performance. If not specified, `OMP_NUM_THREADS` is set to `1` by default. You can change this by using the `--omp_num_threads` argument: +```bash +python3 run_server.py --port 9090 \ + --backend faster_whisper \ + --omp_num_threads 4 +``` ### Running the Client - Initializing the client: diff --git a/run_server.py b/run_server.py index 4c6403b..3caa788 100644 --- a/run_server.py +++ b/run_server.py @@ -1,5 +1,5 @@ import argparse -from whisper_live.server import TranscriptionServer +import os if __name__ == "__main__": parser = argparse.ArgumentParser() @@ -21,12 +21,20 @@ if __name__ == "__main__": parser.add_argument('--trt_multilingual', '-m', action="store_true", help='Boolean only for TensorRT model. True if multilingual.') + parser.add_argument('--omp_num_threads', '-omp', + type=int, + default=1, + help="Number of threads to use for OpenMP") args = parser.parse_args() if args.backend == "tensorrt": if args.trt_model_path is None: raise ValueError("Please Provide a valid tensorrt model path") + if "OMP_NUM_THREADS" not in os.environ: + os.environ["OMP_NUM_THREADS"] = str(args.omp_num_threads) + + from whisper_live.server import TranscriptionServer server = TranscriptionServer() server.run( "0.0.0.0",