initial commit

Browse files

Files changed (7) hide show

.idea/.gitignore +8 -0
README.md +140 -0
config.json +291 -0
handler.py +43 -0
requirements.txt +0 -0
tokenizer.json +0 -0
vocabulary.txt +0 -0

.idea/.gitignore ADDED Viewed

	@@ -0,0 +1,8 @@

+# Default ignored files
+/shelf/
+/workspace.xml
+# Editor-based HTTP Client requests
+/httpRequests/
+# Datasource local storage ignored files
+/dataSources/
+/dataSources.local.xml

README.md ADDED Viewed

	@@ -0,0 +1,140 @@

+---
+language:
+  - en
+  - zh
+  - de
+  - es
+  - ru
+  - ko
+  - fr
+  - ja
+  - pt
+  - tr
+  - pl
+  - ca
+  - nl
+  - ar
+  - sv
+  - it
+  - id
+  - hi
+  - fi
+  - vi
+  - he
+  - uk
+  - el
+  - ms
+  - cs
+  - ro
+  - da
+  - hu
+  - ta
+  - 'no'
+  - th
+  - ur
+  - hr
+  - bg
+  - lt
+  - la
+  - mi
+  - ml
+  - cy
+  - sk
+  - te
+  - fa
+  - lv
+  - bn
+  - sr
+  - az
+  - sl
+  - kn
+  - et
+  - mk
+  - br
+  - eu
+  - is
+  - hy
+  - ne
+  - mn
+  - bs
+  - kk
+  - sq
+  - sw
+  - gl
+  - mr
+  - pa
+  - si
+  - km
+  - sn
+  - yo
+  - so
+  - af
+  - oc
+  - ka
+  - be
+  - tg
+  - sd
+  - gu
+  - am
+  - yi
+  - lo
+  - uz
+  - fo
+  - ht
+  - ps
+  - tk
+  - nn
+  - mt
+  - sa
+  - lb
+  - my
+  - bo
+  - tl
+  - mg
+  - as
+  - tt
+  - haw
+  - ln
+  - ha
+  - ba
+  - jw
+  - su
+tags:
+  - audio
+  - automatic-speech-recognition
+license: mit
+library_name: ctranslate2
+---
+# Whisper large-v2 model for CTranslate2
+This repository contains the conversion of [openai/whisper-large-v2](https://huggingface.co/openai/whisper-large-v2) to the [CTranslate2](https://github.com/OpenNMT/CTranslate2) model format.
+This model can be used in CTranslate2 or projects based on CTranslate2 such as [faster-whisper](https://github.com/guillaumekln/faster-whisper).
+## Example
+```python
+from faster_whisper import WhisperModel
+model = WhisperModel("large-v2")
+segments, info = model.transcribe("audio.mp3")
+for segment in segments:
+    print("[%.2fs -> %.2fs] %s" % (segment.start, segment.end, segment.text))
+```
+## Conversion details
+The original model was converted with the following command:
+```
+ct2-transformers-converter --model openai/whisper-large-v2 --output_dir faster-whisper-large-v2 \
+    --copy_files tokenizer.json --quantization float16
+```
+Note that the model weights are saved in FP16. This type can be changed when the model is loaded using the [`compute_type` option in CTranslate2](https://opennmt.net/CTranslate2/quantization.html).
+## More information
+**For more information about the original model, see its [model card](https://huggingface.co/openai/whisper-large-v2).**

config.json ADDED Viewed

	@@ -0,0 +1,291 @@

+{
+  "alignment_heads": [
+    [
+      10,
+      12
+    ],
+    [
+      13,
+      17
+    ],
+    [
+      16,
+      11
+    ],
+    [
+      16,
+      12
+    ],
+    [
+      16,
+      13
+    ],
+    [
+      17,
+      15
+    ],
+    [
+      17,
+      16
+    ],
+    [
+      18,
+      4
+    ],
+    [
+      18,
+      11
+    ],
+    [
+      18,
+      19
+    ],
+    [
+      19,
+      11
+    ],
+    [
+      21,
+      2
+    ],
+    [
+      21,
+      3
+    ],
+    [
+      22,
+      3
+    ],
+    [
+      22,
+      9
+    ],
+    [
+      22,
+      12
+    ],
+    [
+      23,
+      5
+    ],
+    [
+      23,
+      7
+    ],
+    [
+      23,
+      13
+    ],
+    [
+      25,
+      5
+    ],
+    [
+      26,
+      1
+    ],
+    [
+      26,
+      12
+    ],
+    [
+      27,
+      15
+    ]
+  ],
+  "lang_ids": [
+    50259,
+    50260,
+    50261,
+    50262,
+    50263,
+    50264,
+    50265,
+    50266,
+    50267,
+    50268,
+    50269,
+    50270,
+    50271,
+    50272,
+    50273,
+    50274,
+    50275,
+    50276,
+    50277,
+    50278,
+    50279,
+    50280,
+    50281,
+    50282,
+    50283,
+    50284,
+    50285,
+    50286,
+    50287,
+    50288,
+    50289,
+    50290,
+    50291,
+    50292,
+    50293,
+    50294,
+    50295,
+    50296,
+    50297,
+    50298,
+    50299,
+    50300,
+    50301,
+    50302,
+    50303,
+    50304,
+    50305,
+    50306,
+    50307,
+    50308,
+    50309,
+    50310,
+    50311,
+    50312,
+    50313,
+    50314,
+    50315,
+    50316,
+    50317,
+    50318,
+    50319,
+    50320,
+    50321,
+    50322,
+    50323,
+    50324,
+    50325,
+    50326,
+    50327,
+    50328,
+    50329,
+    50330,
+    50331,
+    50332,
+    50333,
+    50334,
+    50335,
+    50336,
+    50337,
+    50338,
+    50339,
+    50340,
+    50341,
+    50342,
+    50343,
+    50344,
+    50345,
+    50346,
+    50347,
+    50348,
+    50349,
+    50350,
+    50351,
+    50352,
+    50353,
+    50354,
+    50355,
+    50356,
+    50357
+  ],
+  "suppress_ids": [
+    1,
+    2,
+    7,
+    8,
+    9,
+    10,
+    14,
+    25,
+    26,
+    27,
+    28,
+    29,
+    31,
+    58,
+    59,
+    60,
+    61,
+    62,
+    63,
+    90,
+    91,
+    92,
+    93,
+    359,
+    503,
+    522,
+    542,
+    873,
+    893,
+    902,
+    918,
+    922,
+    931,
+    1350,
+    1853,
+    1982,
+    2460,
+    2627,
+    3246,
+    3253,
+    3268,
+    3536,
+    3846,
+    3961,
+    4183,
+    4667,
+    6585,
+    6647,
+    7273,
+    9061,
+    9383,
+    10428,
+    10929,
+    11938,
+    12033,
+    12331,
+    12562,
+    13793,
+    14157,
+    14635,
+    15265,
+    15618,
+    16553,
+    16604,
+    18362,
+    18956,
+    20075,
+    21675,
+    22520,
+    26130,
+    26161,
+    26435,
+    28279,
+    29464,
+    31650,
+    32302,
+    32470,
+    36865,
+    42863,
+    47425,
+    49870,
+    50254,
+    50258,
+    50358,
+    50359,
+    50360,
+    50361,
+    50362
+  ],
+  "suppress_ids_begin": [
+    220,
+    50257
+  ]
+}

handler.py ADDED Viewed

	@@ -0,0 +1,43 @@

+import io
+import base64
+from faster_whisper import WhisperModel
+import logging
+logging.basicConfig(level=logging.DEBUG)
+class EndpointHandler:
+    def __init__(self, path=""):
+        self.model = WhisperModel("large-v2", num_workers=30)
+    def __call__(self, data: dict[str, str]):
+        # process inputs
+        inputs = data.pop("inputs", data)
+        language = data.pop("language", "de")
+        task = data.pop("task", "transcribe")
+        # Decode base64 string to bytes
+        audio_bytes_decoded = base64.b64decode(inputs)
+        logging.debug(f"Decoded Bytes Length: {len(audio_bytes_decoded)}")
+        audio_bytes = io.BytesIO(audio_bytes_decoded)
+        # run inference pipeline
+        logging.info("Running inference...")
+        segments, info = self.model.transcribe(audio_bytes, language=language, task=task)
+        # postprocess the prediction
+        full_text = []
+        for segment in segments:
+            full_text.append({"segmentId": segment.id,
+                              "text": segment.text,
+                              "timestamps": {
+                                  "start": segment.start,
+                                  "end": segment.end
+                              }
+                              })
+            if segment.id % 100 == 0:
+                logging.info("segment " + str(segment.id) + " transcribed")
+        logging.info("Inference completed.")
+        return full_text

requirements.txt ADDED Viewed

Binary file (104 Bytes). View file

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

vocabulary.txt ADDED Viewed

The diff for this file is too large to render. See raw diff