Hide101111001111000
/

llm-jp-3-13b-it_lora-DPO-ja

@@ -21,139 +21,38 @@ This llama model was trained 2x faster with [Unsloth](https://github.com/unsloth
 [<img src="https://raw.githubusercontent.com/unslothai/unsloth/main/images/unsloth%20made%20with%20love.png" width="200"/>](https://github.com/unslothai/unsloth)
-I performed DPO based on the already fine-tuned Hide101111001111000/llm-jp-3-13b-it_lora_3.
 --python--
-!pip install unsloth
-# Also get the latest nightly Unsloth!
-!pip uninstall unsloth -y && pip install --upgrade --no-cache-dir --no-deps git+https://github.com/unslothai/unsloth.git
-from unsloth import PatchDPOTrainer
-PatchDPOTrainer()
-from unsloth import FastLanguageModel
 import torch
-max_seq_length = 2200 # Choose any! We auto support RoPE Scaling internally!
-dtype = None # None for auto detection. Float16 for Tesla T4, V100, Bfloat16 for Ampere+
-load_in_4bit = True # Use 4bit quantization to reduce memory usage. Can be False.
-HF_TOKEN = "your-token"#ご自身のToken
-model, tokenizer = FastLanguageModel.from_pretrained(
-    model_name = "Hide101111001111000/llm-jp-3-13b-it_lora_3.", # 自分がUnslothを使ってFTして、loraだけアップロードしているモデル
-    max_seq_length = max_seq_length,
-    dtype = dtype,
-    load_in_4bit = load_in_4bit,
-    token = HF_TOKEN,
-)
-from huggingface_hub import login
-# 生成したトークンをペースト
-login(HF_TOKEN)
-from datasets import load_dataset
-# データセットをロード
-ds = load_dataset("llm-jp/hh-rlhf-12k-ja")
-#フィルタリング "conversationsの処理"
-def extract_anthropic_prompt(sample):
-    # 'conversations' カラムから最後の 'value' を取得
-    conversations = sample.get("conversations", [])
-    if not conversations or not isinstance(conversations, list):
-        raise KeyError(f"Key 'conversations' not found or is not a list in sample: {sample}")
-    last_conversation = conversations[-1]
-    if 'value' not in last_conversation:
-        raise KeyError(f"Key 'value' not found in last conversation: {last_conversation}")
-    prompt = last_conversation['value']
-    # 'chosen' と 'rejected' フィールドを使用する
-    chosen_text = sample["chosen"].replace("\\n ", "\\n")
-    rejected_text = sample["rejected"].replace("\\n ", "\\n")
-    return {
-        "prompt": prompt,
-        "chosen": chosen_text,
-        "rejected": rejected_text,
-    }
-# フィルタリング関数を定義
-def filter_short_examples(example):
-    return (
-        len(example['prompt']) <= 2000 and
-        len(example['chosen']) <= 2000 and
-        len(example['rejected']) <= 2000
-    )
-# トレーニングデータをフィルタリング
-ds_filter = ds['train'].map(extract_anthropic_prompt)
-filtered_train = ds_filter.filter(filter_short_examples)
-# データセットをトレーニング用と評価用に分割 (80%をトレーニング用、20%を評価用)
-train_size = int(0.8 * len(filtered_train))  # トレーニングデータのサイズ
-eval_size = len(filtered_train) - train_size  # 評価データのサイズ
-# インデックスを順序通りに生成 (ランダム性なし)
-train_indices = list(range(train_size))  # トレーニング用インデックス
-eval_indices = list(range(train_size, len(filtered_train)))  # 評価用インデックス
-# トレーニングデータと評価データを選択
-train_dataset = filtered_train.select(train_indices)
-eval_dataset = filtered_train.select(eval_indices)
-# データセットのサイズを出力
-print(f"トレーニングデータセットのサイズ: {len(train_dataset)}")
-print(f"評価データセットのサイズ: {len(eval_dataset)}")
-use_dataset = train_dataset.select(range(500))
-use_dataset
-# One must patch the DPO Trainer first!
-from unsloth import PatchDPOTrainer
-PatchDPOTrainer()
-from transformers import TrainingArguments
-from trl import DPOTrainer, DPOConfig
-from unsloth import is_bfloat16_supported
-dpo_trainer = DPOTrainer(
-    model = model,
-    ref_model = None,
-    args = DPOConfig(
-        per_device_train_batch_size = 2,
-        gradient_accumulation_steps = 4,
-        warmup_ratio = 0.1,
-        num_train_epochs = 1,
-        learning_rate = 5e-6,
-        fp16 = not is_bfloat16_supported(),
-        bf16 = is_bfloat16_supported(),
-        logging_steps = 1,
-        optim = "adamw_8bit",
-        weight_decay = 0.0,
-        lr_scheduler_type = "linear",
-        seed = 42,
-        output_dir = "outputs",
-        report_to = "none", # Use this for WandB etc
-    ),
-    beta = 0.1,
-    train_dataset = use_dataset, #raw_datasets["train"],
-    # eval_dataset = raw_datasets["test"],
-    tokenizer = tokenizer,
-    max_length = 2048,
-    max_prompt_length = 1024,
-)
-dpo_trainer.train()
-# ELYZA-tasks-100-TVの読み込み。事前にファイルをアップロードしてください
-# データセットの読み込み。
-# omnicampusの開発環境では、左にタスクのjsonlをドラッグアンドドロップしてから実行。
 import json
 datasets = []
-with open("/content/elyza-tasks-100-TV_0.jsonl", "r") as f:
     item = ""
     for line in f:
       line = line.strip()
@@ -162,10 +61,10 @@ with open("/content/elyza-tasks-100-TV_0.jsonl", "r") as f:
         datasets.append(json.loads(item))
         item = ""
-# 学習したモデルを用いてタスクを実行
 from tqdm import tqdm
-# 推論するためにモデルのモードを変更
 FastLanguageModel.for_inference(model)
 results = []
@@ -176,23 +75,12 @@ for dt in tqdm(datasets):
   inputs = tokenizer([prompt], return_tensors = "pt").to(model.device)
-  outputs = model.generate(**inputs, max_new_tokens = 2048, use_cache = True, do_sample=False, repetition_penalty=1.2)
   prediction = tokenizer.decode(outputs[0], skip_special_tokens=True).split('\n### 回答')[-1]
   results.append({"task_id": dt["task_id"], "input": input, "output": prediction})
-# jsonlで保存
-with open(f"llm-jp-3-13b-it_lora-DPO-ja-output.jsonl", 'w', encoding='utf-8') as f:
     for result in results:
         json.dump(result, f, ensure_ascii=False)
-        f.write('\n')
-#modelをhugging faceにアップロード
-# LoRAアダプタだけ保存
-model.push_to_hub_merged(
-    "llm-jp-3-13b-it_lora-DPO-ja",#保存するモデルの名前
-    tokenizer=tokenizer,
-    save_method="lora",#loraだけ保存
-    token=HF_TOKEN,
-    private=True
-)

 [<img src="https://raw.githubusercontent.com/unslothai/unsloth/main/images/unsloth%20made%20with%20love.png" width="200"/>](https://github.com/unslothai/unsloth)
+#推論用コード
+本モデルを用いてELYZA-tasks-100-TVの出力を得るためのコードです。
+このコードを動作させることとで課題として提出可能なjsonlファイル得れるようになっています。
 --python--
+!pip uninstall unsloth -y
+!pip install --upgrade --no-cache-dir "unsloth[colab-new] @ git+https://github.com/unslothai/unsloth.git"
+!pip install --upgrade torch
+!pip install --upgrade xformers
 import torch
+if torch.cuda.get_device_capability()[0] >= 8:
+    !pip install --no-deps packaging ninja einops "flash-attn>=2.6.3"
+from unsloth import FastLanguageModel
+max_seq_length = 512 # unslothではRoPEをサポートしているのでコンテキスト長は自由に設定可能
+dtype = None # Noneにしておけば自動で設定
+load_in_4bit = True # 今回は13Bモデルを扱うためTrue
+model, tokenizer = FastLanguageModel.from_pretrained(
+    model_name="Hide101111001111000/llm-jp-3-13b-it_lora-DPO-ja",
+    dtype=dtype,
+    load_in_4bit=load_in_4bit,
+    trust_remote_code=True,
+)
 import json
 datasets = []
+with open("/content/elyza-tasks-100-TV_0 .jsonl", "r") as f:
     item = ""
     for line in f:
       line = line.strip()
         datasets.append(json.loads(item))
         item = ""
 from tqdm import tqdm
 FastLanguageModel.for_inference(model)
 results = []
   inputs = tokenizer([prompt], return_tensors = "pt").to(model.device)
+  outputs = model.generate(**inputs, max_new_tokens = 512, use_cache = True, do_sample=False, repetition_penalty=1.2)
   prediction = tokenizer.decode(outputs[0], skip_special_tokens=True).split('\n### 回答')[-1]
   results.append({"task_id": dt["task_id"], "input": input, "output": prediction})
+with open(f"model_output.jsonl", 'w', encoding='utf-8') as f:
     for result in results:
         json.dump(result, f, ensure_ascii=False)
+        f.write('\n')