Spaces:

ashourzadeh7
/

nllb-translation-demo-2

Paused

App Files Files Community

ashourzadeh7 commited on Jul 6, 2024

Commit

3ca116b

verified ·

1 Parent(s): a3f20e3

Upload 3 files

Browse files

Files changed (3) hide show

app.py +85 -0
flores200_codes.py +9 -0
requirements.txt +4 -0

app.py ADDED Viewed

	@@ -0,0 +1,85 @@

+import os
+import torch
+import gradio as gr
+import time
+from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, pipeline
+from flores200_codes import flores_codes
+def load_models():
+    # build model and tokenizer
+    model_name_dict = {#'nllb-finetuned-kutofa': 'ashourzadeh7/nllb-finetuned-kutofa',
+                  #'nllb-1.3B': 'facebook/nllb-200-1.3B',
+                  #'nllb-distilled-1.3B': 'facebook/nllb-200-distilled-1.3B',
+                  'nllb-3.3B': 'facebook/nllb-200-3.3B',
+                  }
+    model_dict = {}
+    for call_name, real_name in model_name_dict.items():
+        print('\tLoading model: %s' % call_name)
+        model = AutoModelForSeq2SeqLM.from_pretrained(real_name)
+        tokenizer = AutoTokenizer.from_pretrained(real_name)
+        model_dict[call_name+'_model'] = model
+        model_dict[call_name+'_tokenizer'] = tokenizer
+    return model_dict
+def translation(source, target, text):
+    if len(model_dict) == 2:
+        model_name = 'nllb-3.3B'
+    start_time = time.time()
+    source = flores_codes[source]
+    target = flores_codes[target]
+    model = model_dict[model_name + '_model']
+    tokenizer = model_dict[model_name + '_tokenizer']
+    translator = pipeline('translation', model=model, tokenizer=tokenizer, src_lang=source, tgt_lang=target)
+    output = translator(text, max_length=400)
+    end_time = time.time()
+    output = output[0]['translation_text']
+    result = {'inference_time': end_time - start_time,
+              'source': source,
+              'target': target,
+              'result': output}
+    return result
+if __name__ == '__main__':
+    print('\tinit models')
+    global model_dict
+    model_dict = load_models()
+    # define gradio demo
+    lang_codes = list(flores_codes.keys())
+    #inputs = [gr.inputs.Radio(['nllb-distilled-600M', 'nllb-1.3B', 'nllb-distilled-1.3B'], label='NLLB Model'),
+    inputs = [gr.components.Dropdown(label='Source', choices=lang_codes),
+              gr.components.Dropdown(label='Target', choices=lang_codes),
+              gr.components.Textbox(lines=5, label="Input text"),
+              ]
+    outputs = gr.components.JSON()
+    title = "NLLB distilled 600M demo"
+    demo_status = "Demo is running on CPU"
+    description = f"Details: https://github.com/facebookresearch/fairseq/tree/nllb. {demo_status}"
+    examples = [
+    ['فارسی', 'کردی', 'سلام، حالتون خوبه؟']
+    ]
+    gr.Interface(translation,
+                 inputs,
+                 outputs,
+                 title=title,
+                 description=description,
+                 ).launch()

flores200_codes.py ADDED Viewed

	@@ -0,0 +1,9 @@

+codes_as_string = '''فارسی	pes_Arab
+کردی	ckb_Arab'''
+codes_as_string = codes_as_string.split('\n')
+flores_codes = {}
+for code in codes_as_string:
+    lang, lang_code = code.split('\t')
+    flores_codes[lang] = lang_code

requirements.txt ADDED Viewed

	@@ -0,0 +1,4 @@

+transformers
+gradio
+torch
+httpx==0.24.1