Hide101111001111000
/

llm-jp-3-13b-it_lora-DPO-ja

 --python--
+!pip install unsloth
+# Also get the latest nightly Unsloth!
+!pip uninstall unsloth -y && pip install --upgrade --no-cache-dir --no-deps git+https://github.com/unslothai/unsloth.git
+from unsloth import PatchDPOTrainer
+PatchDPOTrainer()
+from unsloth import FastLanguageModel
+import torch
+max_seq_length = 2200 # Choose any! We auto support RoPE Scaling internally!
+dtype = None # None for auto detection. Float16 for Tesla T4, V100, Bfloat16 for Ampere+
+load_in_4bit = True # Use 4bit quantization to reduce memory usage. Can be False.
+HF_TOKEN = "your-token"#ご自身のToken
+model, tokenizer = FastLanguageModel.from_pretrained(
+    model_name = "Hide101111001111000/llm-jp-3-13b-it_lora_3.", # 自分がUnslothを使ってFTして、loraだけアップロードしているモデル
+    max_seq_length = max_seq_length,
+    dtype = dtype,
+    load_in_4bit = load_in_4bit,
+    token = HF_TOKEN,
+)
+from huggingface_hub import login
+# 生成したトークンをペースト
+login(HF_TOKEN)
+from datasets import load_dataset
+# データセットをロード
+ds = load_dataset("llm-jp/hh-rlhf-12k-ja")
+#フィルタリング "conversationsの処理"
+def extract_anthropic_prompt(sample):
+    # 'conversations' カラムから最後の 'value' を取得
+    conversations = sample.get("conversations", [])
+    if not conversations or not isinstance(conversations, list):
+        raise KeyError(f"Key 'conversations' not found or is not a list in sample: {sample}")
+    last_conversation = conversations[-1]
+    if 'value' not in last_conversation:
+        raise KeyError(f"Key 'value' not found in last conversation: {last_conversation}")
+    prompt = last_conversation['value']
+    # 'chosen' と 'rejected' フィールドを使用する
+    chosen_text = sample["chosen"].replace("\\n ", "\\n")
+    rejected_text = sample["rejected"].replace("\\n ", "\\n")
+    return {
+        "prompt": prompt,
+        "chosen": chosen_text,
+        "rejected": rejected_text,
+    }
+# フィルタリング関数を定義
+def filter_short_examples(example):
+    return (
+        len(example['prompt']) <= 2000 and
+        len(example['chosen']) <= 2000 and
+        len(example['rejected']) <= 2000
+    )
+# トレーニングデータをフィルタリング
+ds_filter = ds['train'].map(extract_anthropic_prompt)
+filtered_train = ds_filter.filter(filter_short_examples)
+# データセットをトレーニング用と評価用に分割 (80%をトレーニング用、20%を評価用)
+train_size = int(0.8 * len(filtered_train))  # トレーニングデータのサイズ
+eval_size = len(filtered_train) - train_size  # 評価データのサイズ
+# インデックスを順序通りに生成 (ランダム性なし)
+train_indices = list(range(train_size))  # トレーニング用インデックス
+eval_indices = list(range(train_size, len(filtered_train)))  # 評価用インデックス
+# トレーニングデータと評価データを選択
+train_dataset = filtered_train.select(train_indices)
+eval_dataset = filtered_train.select(eval_indices)
+# データセットのサイズを出力
+print(f"トレーニングデータセットのサイズ: {len(train_dataset)}")
+print(f"評価データセットのサイズ: {len(eval_dataset)}")
+use_dataset = train_dataset.select(range(500))
+use_dataset
+# One must patch the DPO Trainer first!
+from unsloth import PatchDPOTrainer
+PatchDPOTrainer()
+from transformers import TrainingArguments
+from trl import DPOTrainer, DPOConfig
+from unsloth import is_bfloat16_supported
+dpo_trainer = DPOTrainer(
+    model = model,
+    ref_model = None,
+    args = DPOConfig(
+        per_device_train_batch_size = 2,
+        gradient_accumulation_steps = 4,
+        warmup_ratio = 0.1,
+        num_train_epochs = 1,
+        learning_rate = 5e-6,
+        fp16 = not is_bfloat16_supported(),
+        bf16 = is_bfloat16_supported(),
+        logging_steps = 1,
+        optim = "adamw_8bit",
+        weight_decay = 0.0,
+        lr_scheduler_type = "linear",
+        seed = 42,
+        output_dir = "outputs",
+        report_to = "none", # Use this for WandB etc
+    ),
+    beta = 0.1,
+    train_dataset = use_dataset, #raw_datasets["train"],
+    # eval_dataset = raw_datasets["test"],
+    tokenizer = tokenizer,
+    max_length = 2048,
+    max_prompt_length = 1024,
+)
+dpo_trainer.train()
+# ELYZA-tasks-100-TVの読み込み。事前にファイルをアップロードしてください
+# データセットの読み込み。
+# omnicampusの開発環境では、左にタスクのjsonlをドラッグアンドドロップしてから実行。
+import json
+datasets = []
+with open("/content/elyza-tasks-100-TV_0.jsonl", "r") as f:
+    item = ""
+    for line in f:
+      line = line.strip()
+      item += line
+      if item.endswith("}"):
+        datasets.append(json.loads(item))
+        item = ""
+# 学習したモデルを用いてタスクを実行
+from tqdm import tqdm
+# 推論するためにモデルのモードを変更
+FastLanguageModel.for_inference(model)
+results = []
+for dt in tqdm(datasets):
+  input = dt["input"]
+  prompt = f"""### 指示\n{input}\n### 回答\n"""
+  inputs = tokenizer([prompt], return_tensors = "pt").to(model.device)
+  outputs = model.generate(**inputs, max_new_tokens = 2048, use_cache = True, do_sample=False, repetition_penalty=1.2)
+  prediction = tokenizer.decode(outputs[0], skip_special_tokens=True).split('\n### 回答')[-1]
+  results.append({"task_id": dt["task_id"], "input": input, "output": prediction})
+#modelをhugging faceにアップロード
+# LoRAアダプタだけ保存
+model.push_to_hub_merged(
+    "llm-jp-3-13b-it_lora-DPO-ja",#保存するモデルの名前
+    tokenizer=tokenizer,
+    save_method="lora",#loraだけ保存
+    token=HF_TOKEN,
+    private=True
+)